chen459664 commited on
Commit
41609bb
·
verified ·
1 Parent(s): 665f911

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. lm-evaluation-harness/results/self_attn/self_attn_Llama-2-7b-hf_23_2025-11-27T20-04-34.905255.json +391 -0
  2. lm-evaluation-harness/results/self_attn/self_attn_Llama-2-7b-hf_30_2025-11-27T21-59-39.324442.json +391 -0
  3. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_10_2025-11-28T00-50-52.233394.json +391 -0
  4. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_16_2025-11-28T02-57-18.949335.json +391 -0
  5. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_18_2025-11-28T03-39-31.246223.json +391 -0
  6. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_21_2025-11-28T04-42-35.112741.json +391 -0
  7. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_22_2025-11-28T05-03-28.688835.json +391 -0
  8. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_23_2025-11-28T05-24-44.208704.json +391 -0
  9. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_25_2025-11-28T06-06-40.169344.json +391 -0
  10. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_26_2025-11-28T06-27-28.235781.json +391 -0
  11. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_27_2025-11-28T06-48-16.839289.json +391 -0
  12. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_3_2025-11-27T22-21-04.718480.json +391 -0
  13. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_5_2025-11-27T23-04-05.761374.json +391 -0
  14. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_6_2025-11-27T23-25-42.163366.json +391 -0
  15. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_7_2025-11-27T23-46-57.703939.json +391 -0
  16. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_8_2025-11-28T00-08-17.351955.json +391 -0
  17. lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_9_2025-11-28T00-29-33.271448.json +391 -0
  18. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_0_2025-07-28T23-30-46.784728.json +391 -0
  19. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_10_2025-07-29T10-48-19.416137.json +391 -0
  20. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_12_2025-07-29T11-16-09.019498.json +391 -0
  21. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_13_2025-07-29T11-30-03.468368.json +391 -0
  22. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_14_2025-07-29T11-43-38.761130.json +391 -0
  23. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_15_2025-07-29T11-57-28.892976.json +391 -0
  24. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_16_2025-07-29T14-05-36.368733.json +391 -0
  25. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_17_2025-07-29T14-19-13.430340.json +391 -0
  26. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_18_2025-07-29T14-33-00.978713.json +391 -0
  27. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_19_2025-07-29T14-46-50.976976.json +391 -0
  28. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_1_2025-07-28T23-44-34.920404.json +391 -0
  29. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_20_2025-07-29T15-35-09.420945.json +391 -0
  30. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_21_2025-07-29T15-48-49.608167.json +391 -0
  31. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_22_2025-07-29T16-02-35.978502.json +391 -0
  32. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_23_2025-07-29T16-16-26.879928.json +391 -0
  33. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_2_2025-07-28T23-58-21.862604.json +391 -0
  34. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_3_2025-07-29T00-12-43.282085.json +391 -0
  35. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_4_2025-07-29T00-26-21.368672.json +391 -0
  36. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_5_2025-07-29T00-40-01.583255.json +391 -0
  37. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_6_2025-07-29T00-53-41.766554.json +391 -0
  38. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_7_2025-07-29T01-07-23.277099.json +391 -0
  39. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_8_2025-07-29T01-21-05.686856.json +391 -0
  40. lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_9_2025-07-29T01-34-44.828836.json +391 -0
  41. lm-evaluation-harness/results/sides2middle/test.py +35 -0
  42. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_10_2025-08-02T16-07-50.178079.json +124 -0
  43. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_11_2025-08-02T16-13-27.178206.json +124 -0
  44. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_12_2025-08-02T16-19-04.221323.json +124 -0
  45. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_13_2025-08-02T16-24-39.397729.json +124 -0
  46. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_14_2025-08-02T16-30-14.496427.json +124 -0
  47. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_15_2025-08-02T16-35-48.156593.json +124 -0
  48. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_16_2025-08-02T16-41-27.127665.json +124 -0
  49. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_17_2025-08-02T16-47-06.554030.json +124 -0
  50. lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_18_2025-08-02T16-52-52.835512.json +124 -0
lm-evaluation-harness/results/self_attn/self_attn_Llama-2-7b-hf_23_2025-11-27T20-04-34.905255.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.4069965870307167,
6
+ "acc_stderr,none": 0.014356399418009124,
7
+ "acc_norm,none": 0.44197952218430037,
8
+ "acc_norm_stderr,none": 0.014512682523128342
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.7293771043771043,
13
+ "acc_stderr,none": 0.009116466166403828,
14
+ "acc_norm,none": 0.6957070707070707,
15
+ "acc_norm_stderr,none": 0.009441202922359185
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6929663608562691,
20
+ "acc_stderr,none": 0.008067548709328657
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5343557060346544,
25
+ "acc_stderr,none": 0.004977988452502653,
26
+ "acc_norm,none": 0.7143995220075682,
27
+ "acc_norm_stderr,none": 0.004507768029590039
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.749183895538629,
32
+ "acc_stderr,none": 0.010113869547069044,
33
+ "acc_norm,none": 0.7595212187159956,
34
+ "acc_norm_stderr,none": 0.00997134536465107
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6558800315706393,
39
+ "acc_stderr,none": 0.013352121905005938
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-layer",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764244567.8509402,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-layer",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1149414.184401356,
389
+ "end_time": 1149955.356212361,
390
+ "total_evaluation_time_seconds": "541.1718110051006"
391
+ }
lm-evaluation-harness/results/self_attn/self_attn_Llama-2-7b-hf_30_2025-11-27T21-59-39.324442.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.42235494880546076,
6
+ "acc_stderr,none": 0.014434138713379976,
7
+ "acc_norm,none": 0.44283276450511944,
8
+ "acc_norm_stderr,none": 0.014515573873348899
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.7218013468013468,
13
+ "acc_stderr,none": 0.009195059601583896,
14
+ "acc_norm,none": 0.6978114478114478,
15
+ "acc_norm_stderr,none": 0.009422719042483181
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6981651376146789,
20
+ "acc_stderr,none": 0.00802890499748231
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5355506871141207,
25
+ "acc_stderr,none": 0.004977152746478608,
26
+ "acc_norm,none": 0.7136028679545907,
27
+ "acc_norm_stderr,none": 0.004511533039406165
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7475516866158868,
32
+ "acc_stderr,none": 0.010135665547362362,
33
+ "acc_norm,none": 0.7611534276387377,
34
+ "acc_norm_stderr,none": 0.009948120385337492
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6558800315706393,
39
+ "acc_stderr,none": 0.013352121905005941
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-layer",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764251016.306515,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-layer",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1155860.743708073,
389
+ "end_time": 1156859.775544951,
390
+ "total_evaluation_time_seconds": "999.0318368780427"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_10_2025-11-28T00-50-52.233394.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3873720136518771,
6
+ "acc_stderr,none": 0.01423587248790987,
7
+ "acc_norm,none": 0.4325938566552901,
8
+ "acc_norm_stderr,none": 0.014478005694182524
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6784511784511784,
13
+ "acc_stderr,none": 0.009584091575640616,
14
+ "acc_norm,none": 0.6262626262626263,
15
+ "acc_norm_stderr,none": 0.00992726705825963
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7116207951070337,
20
+ "acc_stderr,none": 0.007923167277795093
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5095598486357299,
25
+ "acc_stderr,none": 0.004988869288786878,
26
+ "acc_norm,none": 0.7016530571599283,
27
+ "acc_norm_stderr,none": 0.00456597493779365
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7480957562568009,
32
+ "acc_stderr,none": 0.01012842133508868,
33
+ "acc_norm,none": 0.7546245919477693,
34
+ "acc_norm_stderr,none": 0.010039831320422396
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6266771902131019,
39
+ "acc_stderr,none": 0.013594002763035512
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764261302.0225787,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1166148.867129788,
389
+ "end_time": 1167132.684549661,
390
+ "total_evaluation_time_seconds": "983.817419872852"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_16_2025-11-28T02-57-18.949335.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3873720136518771,
6
+ "acc_stderr,none": 0.01423587248790987,
7
+ "acc_norm,none": 0.41552901023890787,
8
+ "acc_norm_stderr,none": 0.014401366641216377
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6839225589225589,
13
+ "acc_stderr,none": 0.009540440071928292,
14
+ "acc_norm,none": 0.6275252525252525,
15
+ "acc_norm_stderr,none": 0.009920469215736019
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6981651376146789,
20
+ "acc_stderr,none": 0.008028904997482298
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5053774148575981,
25
+ "acc_stderr,none": 0.004989492828168542,
26
+ "acc_norm,none": 0.7016530571599283,
27
+ "acc_norm_stderr,none": 0.0045659749377936504
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7540805223068553,
32
+ "acc_stderr,none": 0.010047331865625208,
33
+ "acc_norm,none": 0.7633297062023939,
34
+ "acc_norm_stderr,none": 0.009916841655042809
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5872138910812944,
39
+ "acc_stderr,none": 0.0138370606486821
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764268906.6151216,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1173749.845766912,
389
+ "end_time": 1174719.400049518,
390
+ "total_evaluation_time_seconds": "969.5542826061137"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_18_2025-11-28T03-39-31.246223.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.4061433447098976,
6
+ "acc_stderr,none": 0.01435165669009786,
7
+ "acc_norm,none": 0.4334470989761092,
8
+ "acc_norm_stderr,none": 0.014481376224558896
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6868686868686869,
13
+ "acc_stderr,none": 0.009516303879309544,
14
+ "acc_norm,none": 0.622895622895623,
15
+ "acc_norm_stderr,none": 0.009945041946366513
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6709480122324158,
20
+ "acc_stderr,none": 0.008218058611362796
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5117506472814181,
25
+ "acc_stderr,none": 0.004988403265931455,
26
+ "acc_norm,none": 0.7110137422824139,
27
+ "acc_norm_stderr,none": 0.004523651184016192
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7529923830250272,
32
+ "acc_stderr,none": 0.010062268140772612,
33
+ "acc_norm,none": 0.7742110990206746,
34
+ "acc_norm_stderr,none": 0.009754980670917337
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6085240726124704,
39
+ "acc_stderr,none": 0.01371748707129085
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764271438.7982068,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1176284.728511361,
389
+ "end_time": 1177251.696949274,
390
+ "total_evaluation_time_seconds": "966.9684379131068"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_21_2025-11-28T04-42-35.112741.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3822525597269625,
6
+ "acc_stderr,none": 0.014200454049979291,
7
+ "acc_norm,none": 0.41552901023890787,
8
+ "acc_norm_stderr,none": 0.014401366641216386
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.67003367003367,
13
+ "acc_stderr,none": 0.009648311574241036,
14
+ "acc_norm,none": 0.6014309764309764,
15
+ "acc_norm_stderr,none": 0.010046455400477933
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7467889908256881,
20
+ "acc_stderr,none": 0.007605587817314102
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5044811790479984,
25
+ "acc_stderr,none": 0.004989581008163197,
26
+ "acc_norm,none": 0.7040430193188608,
27
+ "acc_norm_stderr,none": 0.00455538837175667
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7388465723612623,
32
+ "acc_stderr,none": 0.010248738649935567,
33
+ "acc_norm,none": 0.7562568008705114,
34
+ "acc_norm_stderr,none": 0.010017199471500612
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6250986582478295,
39
+ "acc_stderr,none": 0.01360554452378801
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764275218.6757095,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1180065.669865653,
389
+ "end_time": 1181035.563834177,
390
+ "total_evaluation_time_seconds": "969.8939685241785"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_22_2025-11-28T05-03-28.688835.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.40017064846416384,
6
+ "acc_stderr,none": 0.014317197787809174,
7
+ "acc_norm,none": 0.4308873720136519,
8
+ "acc_norm_stderr,none": 0.01447113339264246
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6881313131313131,
13
+ "acc_stderr,none": 0.00950582334581766,
14
+ "acc_norm,none": 0.6224747474747475,
15
+ "acc_norm_stderr,none": 0.009947227833469428
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7584097859327217,
20
+ "acc_stderr,none": 0.007486592125266867
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5166301533559052,
25
+ "acc_stderr,none": 0.004987020679861268,
26
+ "acc_norm,none": 0.7164907388966342,
27
+ "acc_norm_stderr,none": 0.00449780302434515
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7513601741022851,
32
+ "acc_stderr,none": 0.010084511234296876,
33
+ "acc_norm,none": 0.7704026115342764,
34
+ "acc_norm_stderr,none": 0.009812682950815197
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6290449881610103,
39
+ "acc_stderr,none": 0.01357639990223157
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764276476.2817006,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1181324.243655074,
389
+ "end_time": 1182289.139969293,
390
+ "total_evaluation_time_seconds": "964.896314219106"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_23_2025-11-28T05-24-44.208704.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3993174061433447,
6
+ "acc_stderr,none": 0.014312094557946702,
7
+ "acc_norm,none": 0.43600682593856654,
8
+ "acc_norm_stderr,none": 0.014491225699230916
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6721380471380471,
13
+ "acc_stderr,none": 0.009632587076170008,
14
+ "acc_norm,none": 0.627104377104377,
15
+ "acc_norm_stderr,none": 0.009922743197129245
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7443425076452599,
20
+ "acc_stderr,none": 0.007629713191771308
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5209121688906593,
25
+ "acc_stderr,none": 0.004985415250690897,
26
+ "acc_norm,none": 0.7199761003784106,
27
+ "acc_norm_stderr,none": 0.004480929450281536
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7453754080522307,
32
+ "acc_stderr,none": 0.010164432237060478,
33
+ "acc_norm,none": 0.7627856365614799,
34
+ "acc_norm_stderr,none": 0.009924694933586345
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6156274664561957,
39
+ "acc_stderr,none": 0.013671567600836192
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764277735.4404094,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1182582.741474708,
389
+ "end_time": 1183564.659762433,
390
+ "total_evaluation_time_seconds": "981.918287724955"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_25_2025-11-28T06-06-40.169344.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.39505119453924914,
6
+ "acc_stderr,none": 0.01428589829293817,
7
+ "acc_norm,none": 0.4257679180887372,
8
+ "acc_norm_stderr,none": 0.014449464278868812
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.686026936026936,
13
+ "acc_stderr,none": 0.00952324533521551,
14
+ "acc_norm,none": 0.6245791245791246,
15
+ "acc_norm_stderr,none": 0.009936218527114302
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7513761467889908,
20
+ "acc_stderr,none": 0.007559492458080537
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.510157339175463,
25
+ "acc_stderr,none": 0.004988751698341148,
26
+ "acc_norm,none": 0.713802031467835,
27
+ "acc_norm_stderr,none": 0.0045105933952898965
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7464635473340587,
32
+ "acc_stderr,none": 0.01015009083455177,
33
+ "acc_norm,none": 0.7676822633297062,
34
+ "acc_norm_stderr,none": 0.009853201384168243
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6156274664561957,
39
+ "acc_stderr,none": 0.013671567600836189
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764280264.8167088,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1185112.433097551,
389
+ "end_time": 1186080.620255082,
390
+ "total_evaluation_time_seconds": "968.1871575308032"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_26_2025-11-28T06-27-28.235781.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.40784982935153585,
6
+ "acc_stderr,none": 0.0143610972884497,
7
+ "acc_norm,none": 0.4257679180887372,
8
+ "acc_norm_stderr,none": 0.014449464278868812
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6847643097643098,
13
+ "acc_stderr,none": 0.009533589368505848,
14
+ "acc_norm,none": 0.6178451178451179,
15
+ "acc_norm_stderr,none": 0.00997074728129243
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7596330275229358,
20
+ "acc_stderr,none": 0.007473634518428277
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5124477195777734,
25
+ "acc_stderr,none": 0.00498823488120674,
26
+ "acc_norm,none": 0.7109141605257917,
27
+ "acc_norm_stderr,none": 0.00452411367125977
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7529923830250272,
32
+ "acc_stderr,none": 0.010062268140772617,
33
+ "acc_norm,none": 0.7758433079434167,
34
+ "acc_norm_stderr,none": 0.009729897956410024
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6235201262825573,
39
+ "acc_stderr,none": 0.01361693196066718
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764281520.770614,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1186369.377036482,
389
+ "end_time": 1187328.686863029,
390
+ "total_evaluation_time_seconds": "959.3098265468143"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_27_2025-11-28T06-48-16.839289.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.33361774744027306,
6
+ "acc_stderr,none": 0.013778687054176534,
7
+ "acc_norm,none": 0.3677474402730375,
8
+ "acc_norm_stderr,none": 0.014090995618168478
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5147306397306397,
13
+ "acc_stderr,none": 0.010255329977562096,
14
+ "acc_norm,none": 0.46380471380471383,
15
+ "acc_norm_stderr,none": 0.010232865550346729
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6406727828746177,
20
+ "acc_stderr,none": 0.008391811770406727
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4053973312089225,
25
+ "acc_stderr,none": 0.004899653704032836,
26
+ "acc_norm,none": 0.5775741884086836,
27
+ "acc_norm_stderr,none": 0.004929361040558203
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6860718171926007,
32
+ "acc_stderr,none": 0.010827928134189644,
33
+ "acc_norm,none": 0.690424374319913,
34
+ "acc_norm_stderr,none": 0.010786656752183345
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5619573796369376,
39
+ "acc_stderr,none": 0.013944181296470804
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764282764.2290099,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1187613.343446134,
389
+ "end_time": 1188577.290310377,
390
+ "total_evaluation_time_seconds": "963.9468642431311"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_3_2025-11-27T22-21-04.718480.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.38139931740614336,
6
+ "acc_stderr,none": 0.01419438908668526,
7
+ "acc_norm,none": 0.42662116040955633,
8
+ "acc_norm_stderr,none": 0.014453185592920293
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6839225589225589,
13
+ "acc_stderr,none": 0.009540440071928283,
14
+ "acc_norm,none": 0.6443602693602694,
15
+ "acc_norm_stderr,none": 0.009822854395535489
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7269113149847095,
20
+ "acc_stderr,none": 0.007792648863012163
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.49751045608444533,
25
+ "acc_stderr,none": 0.004989719559439897,
26
+ "acc_norm,none": 0.6937860983867755,
27
+ "acc_norm_stderr,none": 0.004599776866717495
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7328618063112078,
32
+ "acc_stderr,none": 0.010323440492612435,
33
+ "acc_norm,none": 0.7453754080522307,
34
+ "acc_norm_stderr,none": 0.010164432237060485
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.632991318074191,
39
+ "acc_stderr,none": 0.013546284512919643
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764252307.7038445,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1157156.933162472,
389
+ "end_time": 1158145.169407391,
390
+ "total_evaluation_time_seconds": "988.2362449190114"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_5_2025-11-27T23-04-05.761374.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.4052901023890785,
6
+ "acc_stderr,none": 0.014346869060229327,
7
+ "acc_norm,none": 0.4445392491467577,
8
+ "acc_norm_stderr,none": 0.014521226405627084
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.702020202020202,
13
+ "acc_stderr,none": 0.009385046066694866,
14
+ "acc_norm,none": 0.6346801346801347,
15
+ "acc_norm_stderr,none": 0.009880576614806928
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7581039755351682,
20
+ "acc_stderr,none": 0.007489818475316369
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5195180242979486,
25
+ "acc_stderr,none": 0.004985978214937927,
26
+ "acc_norm,none": 0.7155945030870344,
27
+ "acc_norm_stderr,none": 0.004502088287470071
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7480957562568009,
32
+ "acc_stderr,none": 0.01012842133508868,
33
+ "acc_norm,none": 0.7616974972796517,
34
+ "acc_norm_stderr,none": 0.009940334245876209
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6140489344909235,
39
+ "acc_stderr,none": 0.013682036993397411
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764254888.521918,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1159732.31976151,
389
+ "end_time": 1160726.212483797,
390
+ "total_evaluation_time_seconds": "993.8927222869825"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_6_2025-11-27T23-25-42.163366.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.4044368600682594,
6
+ "acc_stderr,none": 0.014342036483436174,
7
+ "acc_norm,none": 0.4189419795221843,
8
+ "acc_norm_stderr,none": 0.014418106953639015
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6847643097643098,
13
+ "acc_stderr,none": 0.00953358936850585,
14
+ "acc_norm,none": 0.6384680134680135,
15
+ "acc_norm_stderr,none": 0.009858506543162062
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7250764525993884,
20
+ "acc_stderr,none": 0.007808909863706473
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5093606851224856,
25
+ "acc_stderr,none": 0.00498890690130773,
26
+ "acc_norm,none": 0.7054371639115714,
27
+ "acc_norm_stderr,none": 0.004549143750428449
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7453754080522307,
32
+ "acc_stderr,none": 0.010164432237060485,
33
+ "acc_norm,none": 0.7562568008705114,
34
+ "acc_norm_stderr,none": 0.010017199471500614
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6195737963693765,
39
+ "acc_stderr,none": 0.013644727908656831
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764256188.1787307,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1161026.057030449,
389
+ "end_time": 1162022.614438559,
390
+ "total_evaluation_time_seconds": "996.5574081100058"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_7_2025-11-27T23-46-57.703939.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.40017064846416384,
6
+ "acc_stderr,none": 0.014317197787809178,
7
+ "acc_norm,none": 0.44283276450511944,
8
+ "acc_norm_stderr,none": 0.0145155738733489
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6986531986531986,
13
+ "acc_stderr,none": 0.009415259879351618,
14
+ "acc_norm,none": 0.6367845117845118,
15
+ "acc_norm_stderr,none": 0.009868397136118793
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7339449541284404,
20
+ "acc_stderr,none": 0.007728763786791682
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5169288986257717,
25
+ "acc_stderr,none": 0.004986920572284439,
26
+ "acc_norm,none": 0.7131049591714798,
27
+ "acc_norm_stderr,none": 0.004513877465062041
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7562568008705114,
32
+ "acc_stderr,none": 0.010017199471500612,
33
+ "acc_norm,none": 0.7747551686615887,
34
+ "acc_norm_stderr,none": 0.009746643471032147
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6258879242304657,
39
+ "acc_stderr,none": 0.013599792958329821
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764257466.6014485,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1162313.317413699,
389
+ "end_time": 1163298.155078476,
390
+ "total_evaluation_time_seconds": "984.8376647769473"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_8_2025-11-28T00-08-17.351955.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.4069965870307167,
6
+ "acc_stderr,none": 0.01435639941800913,
7
+ "acc_norm,none": 0.44283276450511944,
8
+ "acc_norm_stderr,none": 0.014515573873348895
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.718013468013468,
13
+ "acc_stderr,none": 0.009233124071053643,
14
+ "acc_norm,none": 0.6813973063973064,
15
+ "acc_norm_stderr,none": 0.009560775507673366
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7614678899082569,
20
+ "acc_stderr,none": 0.007454040738031629
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5123481378211512,
25
+ "acc_stderr,none": 0.004988259530472496,
26
+ "acc_norm,none": 0.7112129057956582,
27
+ "acc_norm_stderr,none": 0.0045227254125568975
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7480957562568009,
32
+ "acc_stderr,none": 0.010128421335088681,
33
+ "acc_norm,none": 0.7600652883569097,
34
+ "acc_norm_stderr,none": 0.009963625892809545
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5840568271507498,
39
+ "acc_stderr,none": 0.013852485356798266
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764258742.9430768,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1163589.141038293,
389
+ "end_time": 1164577.802904833,
390
+ "total_evaluation_time_seconds": "988.6618665400892"
391
+ }
lm-evaluation-harness/results/self_attn_Qwen2.5-7B/self_attn_Qwen2.5-7B_9_2025-11-28T00-29-33.271448.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.38310580204778155,
6
+ "acc_stderr,none": 0.014206472661672883,
7
+ "acc_norm,none": 0.4129692832764505,
8
+ "acc_norm_stderr,none": 0.014388344935398326
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6944444444444444,
13
+ "acc_stderr,none": 0.009452181213593472,
14
+ "acc_norm,none": 0.6338383838383839,
15
+ "acc_norm_stderr,none": 0.009885391390947724
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7422018348623853,
20
+ "acc_stderr,none": 0.007650564175824776
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4998008364867556,
25
+ "acc_stderr,none": 0.004989781015595464,
26
+ "acc_norm,none": 0.6946823341963753,
27
+ "acc_norm_stderr,none": 0.004596006250433613
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7421109902067464,
32
+ "acc_stderr,none": 0.010206956662056264,
33
+ "acc_norm,none": 0.7562568008705114,
34
+ "acc_norm_stderr,none": 0.01001719947150061
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6203630623520127,
39
+ "acc_stderr,none": 0.013639245403711166
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/Qwen2.5-7B-quantization-layer"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/Qwen2.5-7B-quantization-layer",
343
+ "model_num_parameters": 7615616512,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "16",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "3761bde4",
360
+ "date": 1764260027.387923,
361
+ "pretty_env_info": "PyTorch version: 2.8.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.10.18 (main, Jun 5 2025, 13:14:17) [GCC 11.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 4\nNUMA node0 CPU(s): 0-15,64-79\nNUMA node1 CPU(s): 16-31,80-95\nNUMA node2 CPU(s): 32-47,96-111\nNUMA node3 CPU(s): 48-63,112-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] numpy==2.2.6\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.10.2.21\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.27.3\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] torch==2.8.0\n[pip3] triton==3.4.0\n[conda] numpy 2.2.6 pypi_0 pypi\n[conda] nvidia-cublas-cu12 12.8.4.1 pypi_0 pypi\n[conda] nvidia-cuda-cupti-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cuda-nvrtc-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-cuda-runtime-cu12 12.8.90 pypi_0 pypi\n[conda] nvidia-cudnn-cu12 9.10.2.21 pypi_0 pypi\n[conda] nvidia-cufft-cu12 11.3.3.83 pypi_0 pypi\n[conda] nvidia-curand-cu12 10.3.9.90 pypi_0 pypi\n[conda] nvidia-cusolver-cu12 11.7.3.90 pypi_0 pypi\n[conda] nvidia-cusparse-cu12 12.5.8.93 pypi_0 pypi\n[conda] nvidia-cusparselt-cu12 0.7.1 pypi_0 pypi\n[conda] nvidia-nccl-cu12 2.27.3 pypi_0 pypi\n[conda] nvidia-nvjitlink-cu12 12.8.93 pypi_0 pypi\n[conda] nvidia-nvtx-cu12 12.8.90 pypi_0 pypi\n[conda] torch 2.8.0 pypi_0 pypi\n[conda] triton 3.4.0 pypi_0 pypi",
362
+ "transformers_version": "4.55.2",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": "3761bde4a46223e738034eac9a2e68a7b5997d5e",
365
+ "tokenizer_pad_token": [
366
+ "<|endoftext|>",
367
+ "151643"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "<|endoftext|>",
371
+ "151643"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ null,
375
+ "None"
376
+ ],
377
+ "eot_token_id": 151643,
378
+ "max_length": 131072,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/Qwen2.5-7B-quantization-layer",
382
+ "model_name_sanitized": "..__models__Qwen2.5-7B-quantization-layer",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 1164871.986465088,
389
+ "end_time": 1165853.722397119,
390
+ "total_evaluation_time_seconds": "981.7359320309479"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_0_2025-07-28T23-30-46.784728.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.310580204778157,
6
+ "acc_stderr,none": 0.013522292098053047,
7
+ "acc_norm,none": 0.3515358361774744,
8
+ "acc_norm_stderr,none": 0.013952413699600938
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5231481481481481,
13
+ "acc_stderr,none": 0.010248782484554474,
14
+ "acc_norm,none": 0.5033670033670034,
15
+ "acc_norm_stderr,none": 0.010259550893798928
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6103975535168196,
20
+ "acc_stderr,none": 0.008529228894936292
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.40460067715594505,
25
+ "acc_stderr,none": 0.0048981151109750345,
26
+ "acc_norm,none": 0.5438159729137622,
27
+ "acc_norm_stderr,none": 0.004970585328297623
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6556039173014145,
32
+ "acc_stderr,none": 0.011086521237125621,
33
+ "acc_norm,none": 0.6610446137105549,
34
+ "acc_norm_stderr,none": 0.011044144419710633
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6148382004735596,
39
+ "acc_stderr,none": 0.013676821287521436
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "24",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753716101.800034,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11966344.959494296,
389
+ "end_time": 11966915.60005682,
390
+ "total_evaluation_time_seconds": "570.6405625231564"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_10_2025-07-29T10-48-19.416137.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.22184300341296928,
6
+ "acc_stderr,none": 0.012141659068147879,
7
+ "acc_norm,none": 0.2841296928327645,
8
+ "acc_norm_stderr,none": 0.013179442447653886
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.26725589225589225,
13
+ "acc_stderr,none": 0.00908046324601747,
14
+ "acc_norm,none": 0.2769360269360269,
15
+ "acc_norm_stderr,none": 0.009182190173795889
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.57217125382263,
20
+ "acc_stderr,none": 0.008653474894637185
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.27394941246763593,
25
+ "acc_stderr,none": 0.004450718673552655,
26
+ "acc_norm,none": 0.30252937661820356,
27
+ "acc_norm_stderr,none": 0.004584144014654945
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5413492927094669,
32
+ "acc_stderr,none": 0.011625864113315818,
33
+ "acc_norm,none": 0.5212187159956474,
34
+ "acc_norm_stderr,none": 0.01165531473228886
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5272296764009471,
39
+ "acc_stderr,none": 0.014031631629827696
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753724398.914574,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11974644.4447293,
389
+ "end_time": 12007568.23148695,
390
+ "total_evaluation_time_seconds": "32923.786757649854"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_12_2025-07-29T11-16-09.019498.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2363481228668942,
6
+ "acc_stderr,none": 0.012414960524301825,
7
+ "acc_norm,none": 0.28242320819112626,
8
+ "acc_norm_stderr,none": 0.013155456884097222
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.29208754208754206,
13
+ "acc_stderr,none": 0.00933070561656908,
14
+ "acc_norm,none": 0.30387205387205385,
15
+ "acc_norm_stderr,none": 0.009437524848293738
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.563914373088685,
20
+ "acc_stderr,none": 0.008673312776324927
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.2916749651463852,
25
+ "acc_stderr,none": 0.0045360453684047146,
26
+ "acc_norm,none": 0.33578968333001397,
27
+ "acc_norm_stderr,none": 0.004713006072807695
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5576713819368879,
32
+ "acc_stderr,none": 0.011587963545507179,
33
+ "acc_norm,none": 0.5174102285092492,
34
+ "acc_norm_stderr,none": 0.011658749823107691
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5422257300710339,
39
+ "acc_stderr,none": 0.01400228450442243
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753758402.1688955,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12008648.559134934,
389
+ "end_time": 12009237.834886754,
390
+ "total_evaluation_time_seconds": "589.2757518198341"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_13_2025-07-29T11-30-03.468368.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2295221843003413,
6
+ "acc_stderr,none": 0.012288926760890788,
7
+ "acc_norm,none": 0.27474402730375425,
8
+ "acc_norm_stderr,none": 0.013044617212771227
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.29419191919191917,
13
+ "acc_stderr,none": 0.009350328648861737,
14
+ "acc_norm,none": 0.29924242424242425,
15
+ "acc_norm_stderr,none": 0.009396447162309822
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5889908256880734,
20
+ "acc_stderr,none": 0.008605429733982185
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.30013941445927106,
25
+ "acc_stderr,none": 0.004573817163007458,
26
+ "acc_norm,none": 0.3482374029077873,
27
+ "acc_norm_stderr,none": 0.004754380554929224
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5609357997823722,
32
+ "acc_stderr,none": 0.011578865649321302,
33
+ "acc_norm,none": 0.5364526659412405,
34
+ "acc_norm_stderr,none": 0.011634779837872287
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5556432517758485,
39
+ "acc_stderr,none": 0.013965196769083558
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753759262.7464747,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12009507.008460265,
389
+ "end_time": 12010072.28374933,
390
+ "total_evaluation_time_seconds": "565.2752890661359"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_14_2025-07-29T11-43-38.761130.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2568259385665529,
6
+ "acc_stderr,none": 0.0127669237941168,
7
+ "acc_norm,none": 0.29692832764505117,
8
+ "acc_norm_stderr,none": 0.01335202597672522
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.37457912457912457,
13
+ "acc_stderr,none": 0.009931758820410624,
14
+ "acc_norm,none": 0.36153198653198654,
15
+ "acc_norm_stderr,none": 0.009858506543162065
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6376146788990825,
20
+ "acc_stderr,none": 0.008407308655864032
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.34007169886476796,
25
+ "acc_stderr,none": 0.0047276480578979305,
26
+ "acc_norm,none": 0.425911173073093,
27
+ "acc_norm_stderr,none": 0.004934698012050245
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5799782372143635,
32
+ "acc_stderr,none": 0.011515615810587486,
33
+ "acc_norm,none": 0.5696409140369967,
34
+ "acc_norm_stderr,none": 0.011552114834700507
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6045777426992897,
39
+ "acc_stderr,none": 0.013741678387545357
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753760086.6636326,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12010332.597327717,
389
+ "end_time": 12010887.576293474,
390
+ "total_evaluation_time_seconds": "554.9789657574147"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_15_2025-07-29T11-57-28.892976.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2645051194539249,
6
+ "acc_stderr,none": 0.012889272949313366,
7
+ "acc_norm,none": 0.30887372013651876,
8
+ "acc_norm_stderr,none": 0.013501770929344004
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.3998316498316498,
13
+ "acc_stderr,none": 0.010051788039412911,
14
+ "acc_norm,none": 0.38552188552188554,
15
+ "acc_norm_stderr,none": 0.009987250004629017
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6498470948012233,
20
+ "acc_stderr,none": 0.008343091327001256
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.367257518422625,
25
+ "acc_stderr,none": 0.004810723108378214,
26
+ "acc_norm,none": 0.47769368651663013,
27
+ "acc_norm_stderr,none": 0.004984813391016204
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6115342763873776,
32
+ "acc_stderr,none": 0.011371877593210252,
33
+ "acc_norm,none": 0.5973884657236126,
34
+ "acc_norm_stderr,none": 0.011442395233488698
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6156274664561957,
39
+ "acc_stderr,none": 0.01367156760083619
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753760909.1441023,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12011154.91440619,
389
+ "end_time": 12011717.708308471,
390
+ "total_evaluation_time_seconds": "562.7939022816718"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_16_2025-07-29T14-05-36.368733.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3174061433447099,
6
+ "acc_stderr,none": 0.01360223908803817,
7
+ "acc_norm,none": 0.3361774744027304,
8
+ "acc_norm_stderr,none": 0.013804855026205761
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.4802188552188552,
13
+ "acc_stderr,none": 0.010251751199542733,
14
+ "acc_norm,none": 0.45496632996632996,
15
+ "acc_norm_stderr,none": 0.01021808445460259
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6691131498470948,
20
+ "acc_stderr,none": 0.008229663469949947
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4206333399721171,
25
+ "acc_stderr,none": 0.004926518439372265,
26
+ "acc_norm,none": 0.5616411073491336,
27
+ "acc_norm_stderr,none": 0.00495171762200797
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6517954298150164,
32
+ "acc_stderr,none": 0.0111152263432444,
33
+ "acc_norm,none": 0.6610446137105549,
34
+ "acc_norm_stderr,none": 0.011044144419710633
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6195737963693765,
39
+ "acc_stderr,none": 0.013644727908656831
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753761732.5349648,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12011977.882450234,
389
+ "end_time": 12019405.184134863,
390
+ "total_evaluation_time_seconds": "7427.301684629172"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_17_2025-07-29T14-19-13.430340.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.3225255972696246,
6
+ "acc_stderr,none": 0.013659980894277378,
7
+ "acc_norm,none": 0.3583617747440273,
8
+ "acc_norm_stderr,none": 0.01401288333485986
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5585016835016835,
13
+ "acc_stderr,none": 0.010189314382749943,
14
+ "acc_norm,none": 0.5505050505050505,
15
+ "acc_norm_stderr,none": 0.010207308833916044
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6474006116207951,
20
+ "acc_stderr,none": 0.00835641249356212
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.45339573790081655,
25
+ "acc_stderr,none": 0.004968058944472163,
26
+ "acc_norm,none": 0.6141206930890261,
27
+ "acc_norm_stderr,none": 0.004858074013443999
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6849836779107725,
32
+ "acc_stderr,none": 0.010838072746240652,
33
+ "acc_norm,none": 0.7034820457018498,
34
+ "acc_norm_stderr,none": 0.010656078922661148
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.632991318074191,
39
+ "acc_stderr,none": 0.013546284512919643
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753769419.7554166,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12019665.750277877,
389
+ "end_time": 12020222.245503832,
390
+ "total_evaluation_time_seconds": "556.4952259548008"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_18_2025-07-29T14-33-00.978713.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.36689419795221845,
6
+ "acc_stderr,none": 0.014084133118104292,
7
+ "acc_norm,none": 0.39078498293515357,
8
+ "acc_norm_stderr,none": 0.014258563880513778
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.656986531986532,
13
+ "acc_stderr,none": 0.009740965666489229,
14
+ "acc_norm,none": 0.6292087542087542,
15
+ "acc_norm_stderr,none": 0.00991129282205692
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6834862385321101,
20
+ "acc_stderr,none": 0.008134928228720325
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.49960167297351127,
25
+ "acc_stderr,none": 0.004989779828043857,
26
+ "acc_norm,none": 0.6684923322047401,
27
+ "acc_norm_stderr,none": 0.004697929774670323
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.719804134929271,
32
+ "acc_stderr,none": 0.01047812201557708,
33
+ "acc_norm,none": 0.7252448313384113,
34
+ "acc_norm_stderr,none": 0.010415033676676042
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6574585635359116,
39
+ "acc_stderr,none": 0.013337483579075923
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753770239.210303,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12020485.407021772,
389
+ "end_time": 12021049.794073468,
390
+ "total_evaluation_time_seconds": "564.3870516959578"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_19_2025-07-29T14-46-50.976976.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.39419795221843,
6
+ "acc_stderr,none": 0.014280522667467325,
7
+ "acc_norm,none": 0.4121160409556314,
8
+ "acc_norm_stderr,none": 0.014383915302225402
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.696969696969697,
13
+ "acc_stderr,none": 0.009430140669278948,
14
+ "acc_norm,none": 0.6658249158249159,
15
+ "acc_norm_stderr,none": 0.009679106032919051
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.7027522935779816,
20
+ "acc_stderr,none": 0.007993793620560269
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.5226050587532364,
25
+ "acc_stderr,none": 0.0049846793593756236,
26
+ "acc_norm,none": 0.6956781517625971,
27
+ "acc_norm_stderr,none": 0.004591792612775578
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7426550598476604,
32
+ "acc_stderr,none": 0.010199921064792512,
33
+ "acc_norm,none": 0.7513601741022851,
34
+ "acc_norm_stderr,none": 0.010084511234296871
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6574585635359116,
39
+ "acc_stderr,none": 0.013337483579075925
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753771064.7664573,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12021311.658574613,
389
+ "end_time": 12021879.792283429,
390
+ "total_evaluation_time_seconds": "568.1337088160217"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_1_2025-07-28T23-44-34.920404.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.36006825938566556,
6
+ "acc_stderr,none": 0.01402751681458519,
7
+ "acc_norm,none": 0.39078498293515357,
8
+ "acc_norm_stderr,none": 0.01425856388051378
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5791245791245792,
13
+ "acc_stderr,none": 0.010130502164066328,
14
+ "acc_norm,none": 0.5694444444444444,
15
+ "acc_norm_stderr,none": 0.010160345396860079
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6253822629969419,
20
+ "acc_stderr,none": 0.008465633983431928
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4420434176458873,
25
+ "acc_stderr,none": 0.0049561470461089675,
26
+ "acc_norm,none": 0.6089424417446724,
27
+ "acc_norm_stderr,none": 0.004869899297734552
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6741022850924918,
32
+ "acc_stderr,none": 0.010935760218903936,
33
+ "acc_norm,none": 0.676822633297062,
34
+ "acc_norm_stderr,none": 0.01091197412428213
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6266771902131019,
39
+ "acc_stderr,none": 0.013594002763035507
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753716937.6073155,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11967180.545716472,
389
+ "end_time": 11967743.735752324,
390
+ "total_evaluation_time_seconds": "563.1900358516723"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_20_2025-07-29T15-35-09.420945.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.34897610921501704,
6
+ "acc_stderr,none": 0.013928933461382504,
7
+ "acc_norm,none": 0.37372013651877134,
8
+ "acc_norm_stderr,none": 0.014137708601759096
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.6359427609427609,
13
+ "acc_stderr,none": 0.009873293392779115,
14
+ "acc_norm,none": 0.5854377104377104,
15
+ "acc_norm_stderr,none": 0.010108889212447769
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6749235474006117,
20
+ "acc_stderr,none": 0.008192427107041335
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4714200358494324,
25
+ "acc_stderr,none": 0.004981623292196195,
26
+ "acc_norm,none": 0.6426010754829715,
27
+ "acc_norm_stderr,none": 0.0047825427541020975
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.7040261153427638,
32
+ "acc_stderr,none": 0.010650414317148126,
33
+ "acc_norm,none": 0.7105549510337323,
34
+ "acc_norm_stderr,none": 0.010581014740675614
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.6069455406471981,
39
+ "acc_stderr,none": 0.013727276249108446
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753773955.6369689,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12024201.714658316,
389
+ "end_time": 12024778.23625948,
390
+ "total_evaluation_time_seconds": "576.5216011628509"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_21_2025-07-29T15-48-49.608167.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.29180887372013653,
6
+ "acc_stderr,none": 0.013284525292403518,
7
+ "acc_norm,none": 0.32764505119453924,
8
+ "acc_norm_stderr,none": 0.01371584794071934
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5614478114478114,
13
+ "acc_stderr,none": 0.01018201027547112,
14
+ "acc_norm,none": 0.5235690235690236,
15
+ "acc_norm_stderr,none": 0.010248378585554035
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6422018348623854,
20
+ "acc_stderr,none": 0.008383924627354822
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4253136825333599,
25
+ "acc_stderr,none": 0.004933800927560535,
26
+ "acc_norm,none": 0.5776737701653057,
27
+ "acc_norm_stderr,none": 0.004929204864315961
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6773667029379761,
32
+ "acc_stderr,none": 0.010907166359856611,
33
+ "acc_norm,none": 0.6882480957562568,
34
+ "acc_norm_stderr,none": 0.010807431424873677
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5643251775848461,
39
+ "acc_stderr,none": 0.01393570973961572
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753774793.0319695,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12025039.384732388,
389
+ "end_time": 12025598.423554692,
390
+ "total_evaluation_time_seconds": "559.0388223044574"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_22_2025-07-29T16-02-35.978502.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.28754266211604096,
6
+ "acc_stderr,none": 0.013226719056266129,
7
+ "acc_norm,none": 0.3412969283276451,
8
+ "acc_norm_stderr,none": 0.013855831287497724
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.5244107744107744,
13
+ "acc_stderr,none": 0.010247548905242264,
14
+ "acc_norm,none": 0.49326599326599324,
15
+ "acc_norm_stderr,none": 0.010258852980991825
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.6247706422018349,
20
+ "acc_stderr,none": 0.008468397820914278
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.4071898028281219,
25
+ "acc_stderr,none": 0.004903066639761952,
26
+ "acc_norm,none": 0.5464050985859391,
27
+ "acc_norm_stderr,none": 0.004968244611429391
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6561479869423286,
32
+ "acc_stderr,none": 0.011082356277961393,
33
+ "acc_norm,none": 0.6610446137105549,
34
+ "acc_norm_stderr,none": 0.011044144419710638
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5769534333070244,
39
+ "acc_stderr,none": 0.013885055359056486
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753775613.2423491,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12025858.877764639,
389
+ "end_time": 12026424.793828467,
390
+ "total_evaluation_time_seconds": "565.9160638283938"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_23_2025-07-29T16-16-26.879928.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.25597269624573377,
6
+ "acc_stderr,none": 0.012753013241244525,
7
+ "acc_norm,none": 0.30802047781569963,
8
+ "acc_norm_stderr,none": 0.01349142951729204
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.4238215488215488,
13
+ "acc_stderr,none": 0.010140006095213603,
14
+ "acc_norm,none": 0.40404040404040403,
15
+ "acc_norm_stderr,none": 0.010069061649549545
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5767584097859327,
20
+ "acc_stderr,none": 0.00864139139911359
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.3657637920732922,
25
+ "acc_stderr,none": 0.004806593424942258,
26
+ "acc_norm,none": 0.47450707030472017,
27
+ "acc_norm_stderr,none": 0.004983291578289042
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.6284004352557128,
32
+ "acc_stderr,none": 0.011274603006724749,
33
+ "acc_norm,none": 0.6218715995647442,
34
+ "acc_norm_stderr,none": 0.011313980666854533
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5461720599842147,
39
+ "acc_stderr,none": 0.013992441563707058
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753776443.2848642,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 12026688.694218436,
389
+ "end_time": 12027255.695248485,
390
+ "total_evaluation_time_seconds": "567.0010300483555"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_2_2025-07-28T23-58-21.862604.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.21843003412969283,
6
+ "acc_stderr,none": 0.012074291605700999,
7
+ "acc_norm,none": 0.2832764505119454,
8
+ "acc_norm_stderr,none": 0.013167478735134576
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2601010101010101,
13
+ "acc_stderr,none": 0.009001718541079954,
14
+ "acc_norm,none": 0.26936026936026936,
15
+ "acc_norm_stderr,none": 0.009103043207756982
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5663608562691131,
20
+ "acc_stderr,none": 0.008667690464344678
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.26199960167297354,
25
+ "acc_stderr,none": 0.0043882375575267285,
26
+ "acc_norm,none": 0.2665803624775941,
27
+ "acc_norm_stderr,none": 0.0044126741709764675
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5424374319912949,
32
+ "acc_stderr,none": 0.011623729421518136,
33
+ "acc_norm,none": 0.49347116430903154,
34
+ "acc_norm_stderr,none": 0.01166482959521097
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5043409629044988,
39
+ "acc_stderr,none": 0.014051956064076908
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753717772.443396,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11968014.83061471,
389
+ "end_time": 11968570.6779589,
390
+ "total_evaluation_time_seconds": "555.8473441898823"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_3_2025-07-29T00-12-43.282085.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2226962457337884,
6
+ "acc_stderr,none": 0.012158314774829941,
7
+ "acc_norm,none": 0.2901023890784983,
8
+ "acc_norm_stderr,none": 0.013261573677520778
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2596801346801347,
13
+ "acc_stderr,none": 0.008996990428562217,
14
+ "acc_norm,none": 0.2727272727272727,
15
+ "acc_norm_stderr,none": 0.00913863072636423
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5406727828746177,
20
+ "acc_stderr,none": 0.008716073497171073
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.2597092212706632,
25
+ "acc_stderr,none": 0.004375788991216849,
26
+ "acc_norm,none": 0.27235610436168095,
27
+ "acc_norm_stderr,none": 0.0044426235908463264
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5380848748639826,
32
+ "acc_stderr,none": 0.011631933367846709,
33
+ "acc_norm,none": 0.49075081610446136,
34
+ "acc_norm_stderr,none": 0.011663828032649183
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5019731649565904,
39
+ "acc_stderr,none": 0.014052376259225636
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753718619.8092546,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11968863.13915367,
389
+ "end_time": 11969432.097406216,
390
+ "total_evaluation_time_seconds": "568.9582525454462"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_4_2025-07-29T00-26-21.368672.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.22440273037542663,
6
+ "acc_stderr,none": 0.01219140493860384,
7
+ "acc_norm,none": 0.2901023890784983,
8
+ "acc_norm_stderr,none": 0.013261573677520778
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.25252525252525254,
13
+ "acc_stderr,none": 0.00891494899149571,
14
+ "acc_norm,none": 0.26725589225589225,
15
+ "acc_norm_stderr,none": 0.00908046324601747
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5,
20
+ "acc_stderr,none": 0.008745054976398168
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.25980880302728543,
25
+ "acc_stderr,none": 0.004376333451909807,
26
+ "acc_norm,none": 0.26867157936666003,
27
+ "acc_norm_stderr,none": 0.004423628080052018
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5386289445048966,
32
+ "acc_stderr,none": 0.011630956681145912,
33
+ "acc_norm,none": 0.5021762785636561,
34
+ "acc_norm_stderr,none": 0.011665713661738868
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5043409629044988,
39
+ "acc_stderr,none": 0.014051956064076903
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753719452.6957104,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11969697.101988131,
389
+ "end_time": 11970250.184012849,
390
+ "total_evaluation_time_seconds": "553.0820247177035"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_5_2025-07-29T00-40-01.583255.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2235494880546075,
6
+ "acc_stderr,none": 0.012174896631202602,
7
+ "acc_norm,none": 0.2909556313993174,
8
+ "acc_norm_stderr,none": 0.013273077865907586
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2516835016835017,
13
+ "acc_stderr,none": 0.008905088235948775,
14
+ "acc_norm,none": 0.26304713804713803,
15
+ "acc_norm_stderr,none": 0.009034514898865824
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.4972477064220184,
20
+ "acc_stderr,none": 0.008744922485713836
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.2588129854610635,
25
+ "acc_stderr,none": 0.004370875625259006,
26
+ "acc_norm,none": 0.2682732523401713,
27
+ "acc_norm_stderr,none": 0.00442155130767846
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5331882480957563,
32
+ "acc_stderr,none": 0.011640096923563131,
33
+ "acc_norm,none": 0.49183895538628947,
34
+ "acc_norm_stderr,none": 0.011664270112244227
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5027624309392266,
39
+ "acc_stderr,none": 0.014052271211616441
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753720271.4643779,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11970515.552060828,
389
+ "end_time": 11971070.398552503,
390
+ "total_evaluation_time_seconds": "554.8464916758239"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_6_2025-07-29T00-53-41.766554.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.22440273037542663,
6
+ "acc_stderr,none": 0.012191404938603838,
7
+ "acc_norm,none": 0.2815699658703072,
8
+ "acc_norm_stderr,none": 0.013143376735009017
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.26052188552188554,
13
+ "acc_stderr,none": 0.009006435890336591,
14
+ "acc_norm,none": 0.2668350168350168,
15
+ "acc_norm_stderr,none": 0.009075915859267255
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5055045871559632,
20
+ "acc_stderr,none": 0.008744525001616654
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.25871340370444135,
25
+ "acc_stderr,none": 0.004370328224831808,
26
+ "acc_norm,none": 0.27325234017128064,
27
+ "acc_norm_stderr,none": 0.004447185883327441
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5402611534276387,
32
+ "acc_stderr,none": 0.011627942981817175,
33
+ "acc_norm,none": 0.49510337323177367,
34
+ "acc_norm_stderr,none": 0.01166526473007814
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5059194948697711,
39
+ "acc_stderr,none": 0.014051500838485807
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753721091.8794072,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11971335.860717384,
389
+ "end_time": 11971890.581561973,
390
+ "total_evaluation_time_seconds": "554.7208445891738"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_7_2025-07-29T01-07-23.277099.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.2295221843003413,
6
+ "acc_stderr,none": 0.012288926760890802,
7
+ "acc_norm,none": 0.27559726962457337,
8
+ "acc_norm_stderr,none": 0.01305716965576184
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2622053872053872,
13
+ "acc_stderr,none": 0.009025197991724842,
14
+ "acc_norm,none": 0.26052188552188554,
15
+ "acc_norm_stderr,none": 0.009006435890336591
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5388379204892967,
20
+ "acc_stderr,none": 0.008718633258803973
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.2590121489743079,
25
+ "acc_stderr,none": 0.00437196954281456,
26
+ "acc_norm,none": 0.27454690300736906,
27
+ "acc_norm_stderr,none": 0.00445373590094782
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5429815016322089,
32
+ "acc_stderr,none": 0.011622641132301952,
33
+ "acc_norm,none": 0.5059847660500544,
34
+ "acc_norm_stderr,none": 0.011664988455853328
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5011838989739542,
39
+ "acc_stderr,none": 0.014052446290529022
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753721910.2163315,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11972155.235194754,
389
+ "end_time": 11972712.09242872,
390
+ "total_evaluation_time_seconds": "556.8572339657694"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_8_2025-07-29T01-21-05.686856.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.23037542662116042,
6
+ "acc_stderr,none": 0.012304928418747611,
7
+ "acc_norm,none": 0.2815699658703072,
8
+ "acc_norm_stderr,none": 0.013143376735009017
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2601010101010101,
13
+ "acc_stderr,none": 0.009001718541079954,
14
+ "acc_norm,none": 0.257996632996633,
15
+ "acc_norm_stderr,none": 0.008977970005203407
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5819571865443425,
20
+ "acc_stderr,none": 0.008626774352070746
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.2656841266679944,
25
+ "acc_stderr,none": 0.004407941058874977,
26
+ "acc_norm,none": 0.2847042421828321,
27
+ "acc_norm_stderr,none": 0.004503511855050033
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5489662676822633,
32
+ "acc_stderr,none": 0.01160974720073308,
33
+ "acc_norm,none": 0.5027203482045702,
34
+ "acc_norm_stderr,none": 0.011665651503000739
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.5130228887134964,
39
+ "acc_stderr,none": 0.014047718393997667
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753722732.614953,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11972977.898943288,
389
+ "end_time": 11973534.502164915,
390
+ "total_evaluation_time_seconds": "556.6032216269523"
391
+ }
lm-evaluation-harness/results/sides2middle/Llama-2-7b-hf-configure_9_2025-07-29T01-34-44.828836.json ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_challenge": {
4
+ "alias": "arc_challenge",
5
+ "acc,none": 0.21928327645051193,
6
+ "acc_stderr,none": 0.012091245787615735,
7
+ "acc_norm,none": 0.2858361774744027,
8
+ "acc_norm_stderr,none": 0.013203196088537362
9
+ },
10
+ "arc_easy": {
11
+ "alias": "arc_easy",
12
+ "acc,none": 0.2647306397306397,
13
+ "acc_stderr,none": 0.009053021086173983,
14
+ "acc_norm,none": 0.265993265993266,
15
+ "acc_norm_stderr,none": 0.00906678956561569
16
+ },
17
+ "boolq": {
18
+ "alias": "boolq",
19
+ "acc,none": 0.5685015290519878,
20
+ "acc_stderr,none": 0.008662594569027304
21
+ },
22
+ "hellaswag": {
23
+ "alias": "hellaswag",
24
+ "acc,none": 0.26767576180043817,
25
+ "acc_stderr,none": 0.00441842761329668,
26
+ "acc_norm,none": 0.2887870942043418,
27
+ "acc_norm_stderr,none": 0.004522725412556948
28
+ },
29
+ "piqa": {
30
+ "alias": "piqa",
31
+ "acc,none": 0.5375408052230686,
32
+ "acc_stderr,none": 0.011632896120570519,
33
+ "acc_norm,none": 0.5043525571273123,
34
+ "acc_norm_stderr,none": 0.011665382144642389
35
+ },
36
+ "winogrande": {
37
+ "alias": "winogrande",
38
+ "acc,none": 0.49171270718232046,
39
+ "acc_stderr,none": 0.014050555322824189
40
+ }
41
+ },
42
+ "group_subtasks": {
43
+ "arc_challenge": [],
44
+ "arc_easy": [],
45
+ "boolq": [],
46
+ "hellaswag": [],
47
+ "piqa": [],
48
+ "winogrande": []
49
+ },
50
+ "configs": {
51
+ "arc_challenge": {
52
+ "task": "arc_challenge",
53
+ "tag": [
54
+ "ai2_arc"
55
+ ],
56
+ "dataset_path": "allenai/ai2_arc",
57
+ "dataset_name": "ARC-Challenge",
58
+ "training_split": "train",
59
+ "validation_split": "validation",
60
+ "test_split": "test",
61
+ "doc_to_text": "Question: {{question}}\nAnswer:",
62
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
63
+ "unsafe_code": false,
64
+ "doc_to_choice": "{{choices.text}}",
65
+ "description": "",
66
+ "target_delimiter": " ",
67
+ "fewshot_delimiter": "\n\n",
68
+ "num_fewshot": 0,
69
+ "metric_list": [
70
+ {
71
+ "metric": "acc",
72
+ "aggregation": "mean",
73
+ "higher_is_better": true
74
+ },
75
+ {
76
+ "metric": "acc_norm",
77
+ "aggregation": "mean",
78
+ "higher_is_better": true
79
+ }
80
+ ],
81
+ "output_type": "multiple_choice",
82
+ "repeats": 1,
83
+ "should_decontaminate": true,
84
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
85
+ "metadata": {
86
+ "version": 1.0,
87
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
88
+ }
89
+ },
90
+ "arc_easy": {
91
+ "task": "arc_easy",
92
+ "tag": [
93
+ "ai2_arc"
94
+ ],
95
+ "dataset_path": "allenai/ai2_arc",
96
+ "dataset_name": "ARC-Easy",
97
+ "training_split": "train",
98
+ "validation_split": "validation",
99
+ "test_split": "test",
100
+ "doc_to_text": "Question: {{question}}\nAnswer:",
101
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
102
+ "unsafe_code": false,
103
+ "doc_to_choice": "{{choices.text}}",
104
+ "description": "",
105
+ "target_delimiter": " ",
106
+ "fewshot_delimiter": "\n\n",
107
+ "num_fewshot": 0,
108
+ "metric_list": [
109
+ {
110
+ "metric": "acc",
111
+ "aggregation": "mean",
112
+ "higher_is_better": true
113
+ },
114
+ {
115
+ "metric": "acc_norm",
116
+ "aggregation": "mean",
117
+ "higher_is_better": true
118
+ }
119
+ ],
120
+ "output_type": "multiple_choice",
121
+ "repeats": 1,
122
+ "should_decontaminate": true,
123
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
124
+ "metadata": {
125
+ "version": 1.0,
126
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
127
+ }
128
+ },
129
+ "boolq": {
130
+ "task": "boolq",
131
+ "tag": [
132
+ "super-glue-lm-eval-v1"
133
+ ],
134
+ "dataset_path": "super_glue",
135
+ "dataset_name": "boolq",
136
+ "training_split": "train",
137
+ "validation_split": "validation",
138
+ "doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
139
+ "doc_to_target": "label",
140
+ "unsafe_code": false,
141
+ "doc_to_choice": [
142
+ "no",
143
+ "yes"
144
+ ],
145
+ "description": "",
146
+ "target_delimiter": " ",
147
+ "fewshot_delimiter": "\n\n",
148
+ "num_fewshot": 0,
149
+ "metric_list": [
150
+ {
151
+ "metric": "acc"
152
+ }
153
+ ],
154
+ "output_type": "multiple_choice",
155
+ "repeats": 1,
156
+ "should_decontaminate": true,
157
+ "doc_to_decontamination_query": "passage",
158
+ "metadata": {
159
+ "version": 2.0,
160
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
161
+ }
162
+ },
163
+ "hellaswag": {
164
+ "task": "hellaswag",
165
+ "tag": [
166
+ "multiple_choice"
167
+ ],
168
+ "dataset_path": "hellaswag",
169
+ "dataset_kwargs": {
170
+ "trust_remote_code": true
171
+ },
172
+ "training_split": "train",
173
+ "validation_split": "validation",
174
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
175
+ "doc_to_text": "{{query}}",
176
+ "doc_to_target": "{{label}}",
177
+ "unsafe_code": false,
178
+ "doc_to_choice": "choices",
179
+ "description": "",
180
+ "target_delimiter": " ",
181
+ "fewshot_delimiter": "\n\n",
182
+ "num_fewshot": 0,
183
+ "metric_list": [
184
+ {
185
+ "metric": "acc",
186
+ "aggregation": "mean",
187
+ "higher_is_better": true
188
+ },
189
+ {
190
+ "metric": "acc_norm",
191
+ "aggregation": "mean",
192
+ "higher_is_better": true
193
+ }
194
+ ],
195
+ "output_type": "multiple_choice",
196
+ "repeats": 1,
197
+ "should_decontaminate": false,
198
+ "metadata": {
199
+ "version": 1.0,
200
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
201
+ }
202
+ },
203
+ "piqa": {
204
+ "task": "piqa",
205
+ "dataset_path": "baber/piqa",
206
+ "dataset_kwargs": {
207
+ "trust_remote_code": true
208
+ },
209
+ "training_split": "train",
210
+ "validation_split": "validation",
211
+ "doc_to_text": "Question: {{goal}}\nAnswer:",
212
+ "doc_to_target": "label",
213
+ "unsafe_code": false,
214
+ "doc_to_choice": "{{[sol1, sol2]}}",
215
+ "description": "",
216
+ "target_delimiter": " ",
217
+ "fewshot_delimiter": "\n\n",
218
+ "num_fewshot": 0,
219
+ "metric_list": [
220
+ {
221
+ "metric": "acc",
222
+ "aggregation": "mean",
223
+ "higher_is_better": true
224
+ },
225
+ {
226
+ "metric": "acc_norm",
227
+ "aggregation": "mean",
228
+ "higher_is_better": true
229
+ }
230
+ ],
231
+ "output_type": "multiple_choice",
232
+ "repeats": 1,
233
+ "should_decontaminate": true,
234
+ "doc_to_decontamination_query": "goal",
235
+ "metadata": {
236
+ "version": 1.0,
237
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
238
+ }
239
+ },
240
+ "winogrande": {
241
+ "task": "winogrande",
242
+ "dataset_path": "winogrande",
243
+ "dataset_name": "winogrande_xl",
244
+ "dataset_kwargs": {
245
+ "trust_remote_code": true
246
+ },
247
+ "training_split": "train",
248
+ "validation_split": "validation",
249
+ "doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
250
+ "doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
251
+ "unsafe_code": false,
252
+ "doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
253
+ "description": "",
254
+ "target_delimiter": " ",
255
+ "fewshot_delimiter": "\n\n",
256
+ "num_fewshot": 0,
257
+ "metric_list": [
258
+ {
259
+ "metric": "acc",
260
+ "aggregation": "mean",
261
+ "higher_is_better": true
262
+ }
263
+ ],
264
+ "output_type": "multiple_choice",
265
+ "repeats": 1,
266
+ "should_decontaminate": true,
267
+ "doc_to_decontamination_query": "sentence",
268
+ "metadata": {
269
+ "version": 1.0,
270
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization"
271
+ }
272
+ }
273
+ },
274
+ "versions": {
275
+ "arc_challenge": 1.0,
276
+ "arc_easy": 1.0,
277
+ "boolq": 2.0,
278
+ "hellaswag": 1.0,
279
+ "piqa": 1.0,
280
+ "winogrande": 1.0
281
+ },
282
+ "n-shot": {
283
+ "arc_challenge": 0,
284
+ "arc_easy": 0,
285
+ "boolq": 0,
286
+ "hellaswag": 0,
287
+ "piqa": 0,
288
+ "winogrande": 0
289
+ },
290
+ "higher_is_better": {
291
+ "arc_challenge": {
292
+ "acc": true,
293
+ "acc_norm": true
294
+ },
295
+ "arc_easy": {
296
+ "acc": true,
297
+ "acc_norm": true
298
+ },
299
+ "boolq": {
300
+ "acc": true
301
+ },
302
+ "hellaswag": {
303
+ "acc": true,
304
+ "acc_norm": true
305
+ },
306
+ "piqa": {
307
+ "acc": true,
308
+ "acc_norm": true
309
+ },
310
+ "winogrande": {
311
+ "acc": true
312
+ }
313
+ },
314
+ "n-samples": {
315
+ "winogrande": {
316
+ "original": 1267,
317
+ "effective": 1267
318
+ },
319
+ "piqa": {
320
+ "original": 1838,
321
+ "effective": 1838
322
+ },
323
+ "hellaswag": {
324
+ "original": 10042,
325
+ "effective": 10042
326
+ },
327
+ "boolq": {
328
+ "original": 3270,
329
+ "effective": 3270
330
+ },
331
+ "arc_easy": {
332
+ "original": 2376,
333
+ "effective": 2376
334
+ },
335
+ "arc_challenge": {
336
+ "original": 1172,
337
+ "effective": 1172
338
+ }
339
+ },
340
+ "config": {
341
+ "model": "hf",
342
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization",
343
+ "model_num_parameters": 6738415616,
344
+ "model_dtype": "torch.float16",
345
+ "model_revision": "main",
346
+ "model_sha": "",
347
+ "batch_size": "32",
348
+ "batch_sizes": [],
349
+ "device": "cuda:0",
350
+ "use_cache": null,
351
+ "limit": null,
352
+ "bootstrap_iters": 100000,
353
+ "gen_kwargs": null,
354
+ "random_seed": 0,
355
+ "numpy_seed": 1234,
356
+ "torch_seed": 1234,
357
+ "fewshot_seed": 1234
358
+ },
359
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
360
+ "date": 1753723551.953976,
361
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: GPU 0: NVIDIA A100-SXM4-80GB\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
362
+ "transformers_version": "4.54.0",
363
+ "lm_eval_version": "0.4.8",
364
+ "upper_git_hash": null,
365
+ "tokenizer_pad_token": [
366
+ "<unk>",
367
+ "0"
368
+ ],
369
+ "tokenizer_eos_token": [
370
+ "</s>",
371
+ "2"
372
+ ],
373
+ "tokenizer_bos_token": [
374
+ "<s>",
375
+ "1"
376
+ ],
377
+ "eot_token_id": 2,
378
+ "max_length": 4096,
379
+ "task_hashes": {},
380
+ "model_source": "hf",
381
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization",
382
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization",
383
+ "system_instruction": null,
384
+ "system_instruction_sha": null,
385
+ "fewshot_as_multiturn": false,
386
+ "chat_template": null,
387
+ "chat_template_sha": null,
388
+ "start_time": 11973797.395326272,
389
+ "end_time": 11974353.644153727,
390
+ "total_evaluation_time_seconds": "556.2488274555653"
391
+ }
lm-evaluation-harness/results/sides2middle/test.py ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import json
3
+ import numpy as np
4
+
5
+ paths = os.listdir('./')
6
+ configure_paths = os.listdir("/mnt/bn/life-mllm/users/cxr/quantization/quantization_metric/bit_layers/sides2middle")
7
+ configure_dir = "/mnt/bn/life-mllm/users/cxr/quantization/quantization_metric/bit_layers/sides2middle"
8
+ tasks = ['piqa', 'winogrande', 'arc_easy', 'arc_challenge', 'hellaswag', 'boolq']
9
+ scores = {}
10
+ for configure_path in configure_paths:
11
+ if 'json' not in configure_path:
12
+ continue
13
+ with open(os.path.join(configure_dir, configure_path), 'r', encoding='utf-8') as f:
14
+ configure_data = json.load(f)
15
+ configure_name = configure_path.split('.')[0]
16
+
17
+ configure_name = configure_name + '_'
18
+ print(configure_name)
19
+ for path in paths:
20
+ if configure_name in path:
21
+ with open(path, 'r', encoding='utf-8') as f:
22
+ data = json.load(f)
23
+ score = []
24
+ # for task, result in data['results'].items():
25
+ # score += result['acc,none']
26
+ # score /= len(data['results'])
27
+ for task in tasks:
28
+ score.append(round(100 * data['results'][task]['acc,none'], 2))
29
+
30
+ scores[','.join([str(x) for x in configure_data])] = score
31
+
32
+ sorted_scores = sorted(scores.items(), key=lambda x: x[1], reverse=False)
33
+ for path, score in sorted_scores:
34
+ print(f"{path}: {score}")
35
+
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_10_2025-08-02T16-07-50.178079.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6902356902356902,
6
+ "acc_stderr,none": 0.00948817285190372,
7
+ "acc_norm,none": 0.6422558922558923,
8
+ "acc_norm_stderr,none": 0.00983577275734336
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754122010.5355117,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13604296.084093029,
122
+ "end_time": 13604376.605857104,
123
+ "total_evaluation_time_seconds": "80.52176407538354"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_11_2025-08-02T16-13-27.178206.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6856060606060606,
6
+ "acc_stderr,none": 0.009526702423162905,
7
+ "acc_norm,none": 0.6414141414141414,
8
+ "acc_norm_stderr,none": 0.009840882301225297
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754122352.7991495,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13604638.541230446,
122
+ "end_time": 13604713.6059981,
123
+ "total_evaluation_time_seconds": "75.06476765498519"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_12_2025-08-02T16-19-04.221323.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.7011784511784511,
6
+ "acc_stderr,none": 0.009392656275408726,
7
+ "acc_norm,none": 0.6666666666666666,
8
+ "acc_norm_stderr,none": 0.009673016668133388
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754122689.2152317,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13604974.937487489,
122
+ "end_time": 13605050.649149235,
123
+ "total_evaluation_time_seconds": "75.71166174672544"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_13_2025-08-02T16-24-39.397729.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.7028619528619529,
6
+ "acc_stderr,none": 0.009377397867796849,
7
+ "acc_norm,none": 0.6590909090909091,
8
+ "acc_norm_stderr,none": 0.009726579593424019
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754123024.9900236,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13605310.644287543,
122
+ "end_time": 13605385.825528724,
123
+ "total_evaluation_time_seconds": "75.18124118074775"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_14_2025-08-02T16-30-14.496427.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6784511784511784,
6
+ "acc_stderr,none": 0.009584091575640623,
7
+ "acc_norm,none": 0.6283670033670034,
8
+ "acc_norm_stderr,none": 0.009915897123658793
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754123360.6299095,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13605646.640681462,
122
+ "end_time": 13605720.924166111,
123
+ "total_evaluation_time_seconds": "74.28348464891315"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_15_2025-08-02T16-35-48.156593.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6944444444444444,
6
+ "acc_stderr,none": 0.009452181213593466,
7
+ "acc_norm,none": 0.6536195286195287,
8
+ "acc_norm_stderr,none": 0.009763542075695734
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754123694.263111,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13605980.79580643,
122
+ "end_time": 13606054.584119977,
123
+ "total_evaluation_time_seconds": "73.7883135471493"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_16_2025-08-02T16-41-27.127665.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6788720538720538,
6
+ "acc_stderr,none": 0.009580787536986797,
7
+ "acc_norm,none": 0.632996632996633,
8
+ "acc_norm_stderr,none": 0.009890173658452121
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754124033.7091055,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13606319.211981172,
122
+ "end_time": 13606393.555451136,
123
+ "total_evaluation_time_seconds": "74.34346996434033"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_17_2025-08-02T16-47-06.554030.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.696969696969697,
6
+ "acc_stderr,none": 0.009430140669278955,
7
+ "acc_norm,none": 0.6641414141414141,
8
+ "acc_norm_stderr,none": 0.009691180932083508
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754124372.8033867,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 86%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13606657.934220264,
122
+ "end_time": 13606732.981796931,
123
+ "total_evaluation_time_seconds": "75.04757666774094"
124
+ }
lm-evaluation-harness/results/singletask_arc_easy/bits_1/Llama-2-7b-hf-configure_18_2025-08-02T16-52-52.835512.json ADDED
@@ -0,0 +1,124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "arc_easy": {
4
+ "alias": "arc_easy",
5
+ "acc,none": 0.6910774410774411,
6
+ "acc_stderr,none": 0.00948104838776135,
7
+ "acc_norm,none": 0.6401515151515151,
8
+ "acc_norm_stderr,none": 0.009848484848484834
9
+ }
10
+ },
11
+ "group_subtasks": {
12
+ "arc_easy": []
13
+ },
14
+ "configs": {
15
+ "arc_easy": {
16
+ "task": "arc_easy",
17
+ "tag": [
18
+ "ai2_arc"
19
+ ],
20
+ "dataset_path": "allenai/ai2_arc",
21
+ "dataset_name": "ARC-Easy",
22
+ "training_split": "train",
23
+ "validation_split": "validation",
24
+ "test_split": "test",
25
+ "doc_to_text": "Question: {{question}}\nAnswer:",
26
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
27
+ "unsafe_code": false,
28
+ "doc_to_choice": "{{choices.text}}",
29
+ "description": "",
30
+ "target_delimiter": " ",
31
+ "fewshot_delimiter": "\n\n",
32
+ "num_fewshot": 0,
33
+ "metric_list": [
34
+ {
35
+ "metric": "acc",
36
+ "aggregation": "mean",
37
+ "higher_is_better": true
38
+ },
39
+ {
40
+ "metric": "acc_norm",
41
+ "aggregation": "mean",
42
+ "higher_is_better": true
43
+ }
44
+ ],
45
+ "output_type": "multiple_choice",
46
+ "repeats": 1,
47
+ "should_decontaminate": true,
48
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
49
+ "metadata": {
50
+ "version": 1.0,
51
+ "pretrained": "../models/patch/Llama-2-7b-hf-quantization-singletask"
52
+ }
53
+ }
54
+ },
55
+ "versions": {
56
+ "arc_easy": 1.0
57
+ },
58
+ "n-shot": {
59
+ "arc_easy": 0
60
+ },
61
+ "higher_is_better": {
62
+ "arc_easy": {
63
+ "acc": true,
64
+ "acc_norm": true
65
+ }
66
+ },
67
+ "n-samples": {
68
+ "arc_easy": {
69
+ "original": 2376,
70
+ "effective": 2376
71
+ }
72
+ },
73
+ "config": {
74
+ "model": "hf",
75
+ "model_args": "pretrained=../models/patch/Llama-2-7b-hf-quantization-singletask",
76
+ "model_num_parameters": 6738415616,
77
+ "model_dtype": "torch.float16",
78
+ "model_revision": "main",
79
+ "model_sha": "",
80
+ "batch_size": "32",
81
+ "batch_sizes": [],
82
+ "device": "cuda:0",
83
+ "use_cache": null,
84
+ "limit": null,
85
+ "bootstrap_iters": 100000,
86
+ "gen_kwargs": null,
87
+ "random_seed": 0,
88
+ "numpy_seed": 1234,
89
+ "torch_seed": 1234,
90
+ "fewshot_seed": 1234
91
+ },
92
+ "git_hash": "d09e03dd14e94a967d3411390961651ffb0dd2f2",
93
+ "date": 1754124716.1872811,
94
+ "pretty_env_info": "PyTorch version: 2.5.1\nIs debug build: False\nCUDA used to build PyTorch: 12.4\nROCM used to build PyTorch: N/A\n\nOS: Debian GNU/Linux 12 (bookworm) (x86_64)\nGCC version: (Debian 12.2.0-14) 12.2.0\nClang version: Could not collect\nCMake version: version 3.25.1\nLibc version: glibc-2.36\n\nPython version: 3.11.2 (main, May 2 2024, 11:59:08) [GCC 12.2.0] (64-bit runtime)\nPython platform: Linux-5.4.143.bsk.7-amd64-x86_64-with-glibc2.36\nIs CUDA available: True\nCUDA runtime version: 12.4.131\nCUDA_MODULE_LOADING set to: LAZY\nGPU models and configuration: \nGPU 0: NVIDIA A100-SXM4-80GB\nGPU 1: NVIDIA A100-SXM4-80GB\n\nNvidia driver version: 535.161.08\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.4.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.4.0\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 57 bits virtual\nByte Order: Little Endian\nCPU(s): 128\nOn-line CPU(s) list: 0-127\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) Platinum 8336C CPU @ 2.30GHz\nCPU family: 6\nModel: 106\nThread(s) per core: 2\nCore(s) per socket: 32\nSocket(s): 2\nStepping: 6\nCPU(s) scaling MHz: 85%\nCPU max MHz: 3500.0000\nCPU min MHz: 800.0000\nBogoMIPS: 4600.00\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush dts acpi mmx fxsr sse sse2 ss ht tm pbe syscall nx pdpe1gb rdtscp lm constant_tsc art arch_perfmon pebs bts rep_good nopl xtopology nonstop_tsc cpuid aperfmperf pni pclmulqdq dtes64 ds_cpl vmx smx est tm2 ssse3 sdbg fma cx16 xtpr pdcm pcid dca sse4_1 sse4_2 x2apic movbe popcnt tsc_deadline_timer aes xsave avx f16c rdrand lahf_lm abm 3dnowprefetch cpuid_fault epb cat_l3 invpcid_single ssbd mba ibrs ibpb stibp ibrs_enhanced tpr_shadow vnmi flexpriority ept vpid ept_ad fsgsbase tsc_adjust bmi1 avx2 smep bmi2 erms invpcid cqm rdt_a avx512f avx512dq rdseed adx smap avx512ifma clflushopt clwb intel_pt avx512cd sha_ni avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves cqm_llc cqm_occup_llc cqm_mbm_total cqm_mbm_local wbnoinvd dtherm ida arat pln pts hwp hwp_act_window hwp_epp hwp_pkg_req avx512vbmi umip pku ospke avx512_vbmi2 gfni vaes vpclmulqdq avx512_vnni avx512_bitalg tme avx512_vpopcntdq rdpid md_clear pconfig flush_l1d arch_capabilities\nVirtualization: VT-x\nL1d cache: 3 MiB (64 instances)\nL1i cache: 2 MiB (64 instances)\nL2 cache: 80 MiB (64 instances)\nL3 cache: 108 MiB (2 instances)\nNUMA node(s): 2\nNUMA node0 CPU(s): 0-31,64-95\nNUMA node1 CPU(s): 32-63,96-127\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Not affected\nVulnerability Mds: Not affected\nVulnerability Meltdown: Not affected\nVulnerability Spec store bypass: Mitigation; Speculative Store Bypass disabled via prctl and seccomp\nVulnerability Spectre v1: Mitigation; usercopy/swapgs barriers and __user pointer sanitization\nVulnerability Spectre v2: Mitigation; Enhanced IBRS, IBPB conditional, RSB filling\nVulnerability Srbds: Not affected\nVulnerability Tsx async abort: Not affected\n\nVersions of relevant libraries:\n[pip3] byted-torch==2.5.1.post1\n[pip3] numpy==1.26.4\n[pip3] torch==2.5.1\n[pip3] torchaudio==2.5.1+cu124\n[pip3] torchvision==0.20.1+cu124\n[pip3] triton==3.1.0\n[conda] Could not collect",
95
+ "transformers_version": "4.54.1",
96
+ "lm_eval_version": "0.4.8",
97
+ "upper_git_hash": null,
98
+ "tokenizer_pad_token": [
99
+ "<unk>",
100
+ "0"
101
+ ],
102
+ "tokenizer_eos_token": [
103
+ "</s>",
104
+ "2"
105
+ ],
106
+ "tokenizer_bos_token": [
107
+ "<s>",
108
+ "1"
109
+ ],
110
+ "eot_token_id": 2,
111
+ "max_length": 4096,
112
+ "task_hashes": {},
113
+ "model_source": "hf",
114
+ "model_name": "../models/patch/Llama-2-7b-hf-quantization-singletask",
115
+ "model_name_sanitized": "..__models__patch__Llama-2-7b-hf-quantization-singletask",
116
+ "system_instruction": null,
117
+ "system_instruction_sha": null,
118
+ "fewshot_as_multiturn": false,
119
+ "chat_template": null,
120
+ "chat_template_sha": null,
121
+ "start_time": 13606999.845416533,
122
+ "end_time": 13607079.263136264,
123
+ "total_evaluation_time_seconds": "79.41771973110735"
124
+ }