{ "comparisons": { "atom-3-4m__gpt-s-1-4m": { "ci": [ 1.0725360001654316, 2.9948782002109477 ], "difference": 2.0326431301418864, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__atom-3-4m": { "ci": [ 14.14042731298988, 16.348961152329448 ], "difference": 15.263255663614302, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__falcon-h1-tiny-r-90m": { "ci": [ 7.754579096585064, 9.903861793761253 ], "difference": 8.820217466378272, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__gpt-s-1-4m": { "ci": [ 16.129475466747913, 18.474038330502612 ], "difference": 17.295898793756187, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__gpt-s2-5m": { "ci": [ 9.256397771070153, 11.38004880060208 ], "difference": 10.331448058039745, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__gpt2-124m": { "ci": [ 0.7505292033322868, 2.7143772000682045 ], "difference": 1.7402333545016724, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__nanowhale-100m-base": { "ci": [ 10.704834399063397, 12.75788043869504 ], "difference": 11.742736810760391, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__pythia-160m": { "ci": [ 8.298184537762872, 10.43661712641525 ], "difference": 9.394030404387307, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__pythia-31m": { "ci": [ 11.57570162659012, 13.766851356448733 ], "difference": 12.659259145773355, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__slm-10m": { "ci": [ 8.323210002923028, 10.392029505159467 ], "difference": 9.360557666936613, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__supra-50m-base": { "ci": [ 1.3509633161095662, 3.2857258513374914 ], "difference": 2.3278594898349447, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "baguettotron__veyra2-apricot-50m-base": { "ci": [ 2.5296915340167394, 4.5001965180924905 ], "difference": 3.547305330072205, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__atom-3-4m": { "ci": [ 5.420215532883351, 7.444210689454811 ], "difference": 6.44303819723603, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__gpt-s-1-4m": { "ci": [ 7.39930477778452, 9.52758940136927 ], "difference": 8.475681327377918, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__gpt-s2-5m": { "ci": [ 0.5031066010624538, 2.4698466105128163 ], "difference": 1.5112305916614743, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__nanowhale-100m-base": { "ci": [ 1.8741543839787467, 3.9548193375016245 ], "difference": 2.922519344382119, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__pythia-160m": { "ci": [ -0.580794542209572, 1.6693586430525291 ], "difference": 0.5738129380090358, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "falcon-h1-tiny-r-90m__pythia-31m": { "ci": [ 2.730495758942819, 4.922286953231492 ], "difference": 3.8390416793950854, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "falcon-h1-tiny-r-90m__slm-10m": { "ci": [ -0.4400500092347234, 1.5559306125986496 ], "difference": 0.5403402005583414, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "gemma-3-270m__atom-3-4m": { "ci": [ 23.385199210048242, 25.661827101907907 ], "difference": 24.542681336414624, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__baguettotron": { "ci": [ 8.231872620662342, 10.330917113466759 ], "difference": 9.279425672800324, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__falcon-h1-tiny-r-90m": { "ci": [ 16.984766782761056, 19.19024856397875 ], "difference": 18.099643139178596, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__gpt-s-1-4m": { "ci": [ 25.31383386993303, 27.774047690234024 ], "difference": 26.57532446655651, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__gpt-s2-5m": { "ci": [ 18.451728069080684, 20.73233091916342 ], "difference": 19.61087373084007, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__gpt-x2-125m": { "ci": [ 2.5100180469723314, 4.277907176531071 ], "difference": 3.395722404973594, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__gpt2-124m": { "ci": [ 10.033805987347693, 12.02164981680341 ], "difference": 11.019659027301998, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__mobilellm-r1-140m-base": { "ci": [ 6.7285212269252535, 8.654045161821886 ], "difference": 7.708492950574182, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__nanowhale-100m-base": { "ci": [ 19.911079186076847, 22.136911037973743 ], "difference": 21.022162483560713, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__pythia-160m": { "ci": [ 17.49071148889552, 19.77861550252012 ], "difference": 18.67345607718763, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__pythia-31m": { "ci": [ 20.72644303179628, 23.094161479902546 ], "difference": 21.93868481857368, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__slm-10m": { "ci": [ 17.546740227831336, 19.736908231495857 ], "difference": 18.639983339736936, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__smollm2-135m": { "ci": [ 0.16836524011662418, 1.9651071736191295 ], "difference": 1.0576894080699517, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__supra-50m-base": { "ci": [ 10.628267365916303, 12.551452199235303 ], "difference": 11.60728516263527, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gemma-3-270m__veyra2-apricot-50m-base": { "ci": [ 11.848671246541473, 13.81413489824284 ], "difference": 12.82673100287253, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-s2-5m__atom-3-4m": { "ci": [ 4.048447096687848, 5.816152073030026 ], "difference": 4.931807605574556, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-s2-5m__gpt-s-1-4m": { "ci": [ 5.994355598920947, 7.87200612832699 ], "difference": 6.964450735716443, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-s2-5m__nanowhale-100m-base": { "ci": [ 0.48907344006903475, 2.3424471779865126 ], "difference": 1.4112887527206444, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-s2-5m__pythia-31m": { "ci": [ 1.3601371630505645, 3.270819706973676 ], "difference": 2.3278110877336107, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__atom-3-4m": { "ci": [ 20.055298011374177, 22.273700065241986 ], "difference": 21.146958931441034, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__baguettotron": { "ci": [ 4.919037209936775, 6.853051793908139 ], "difference": 5.883703267826731, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__falcon-h1-tiny-r-90m": { "ci": [ 13.662735239486574, 15.718124447804271 ], "difference": 14.703920734205003, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__gpt-s-1-4m": { "ci": [ 22.00694789823436, 24.304170912361922 ], "difference": 23.17960206158292, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__gpt-s2-5m": { "ci": [ 15.169372508200368, 17.236224767654043 ], "difference": 16.215151325866476, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__gpt2-124m": { "ci": [ 6.747015411597489, 8.522372794247302 ], "difference": 7.623936622328404, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__mobilellm-r1-140m-base": { "ci": [ 3.4001537890801874, 5.2268186213654895 ], "difference": 4.31277054560059, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__nanowhale-100m-base": { "ci": [ 16.613503795581103, 18.6676187860639 ], "difference": 17.626440078587123, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__pythia-160m": { "ci": [ 14.206915778777304, 16.37469514048295 ], "difference": 15.277733672214039, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__pythia-31m": { "ci": [ 17.423412908889823, 19.65878482694102 ], "difference": 18.542962413600087, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__slm-10m": { "ci": [ 14.224111495050982, 16.29911483056891 ], "difference": 15.244260934763345, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__supra-50m-base": { "ci": [ 7.358523605020158, 9.038699445436707 ], "difference": 8.211562757661675, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt-x2-125m__veyra2-apricot-50m-base": { "ci": [ 8.539745385852653, 10.31133646972243 ], "difference": 9.431008597898938, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__atom-3-4m": { "ci": [ 12.414672706067238, 14.51910829641859 ], "difference": 13.523022309112628, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__falcon-h1-tiny-r-90m": { "ci": [ 6.088666706845199, 8.086574541183792 ], "difference": 7.079984111876598, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__gpt-s-1-4m": { "ci": [ 14.470574653613818, 16.628850295397264 ], "difference": 15.555665439254517, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__gpt-s2-5m": { "ci": [ 7.590743978294489, 9.543732770147578 ], "difference": 8.591214703538071, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__nanowhale-100m-base": { "ci": [ 9.06687024657007, 10.940810819228764 ], "difference": 10.002503456258717, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__pythia-160m": { "ci": [ 6.665984868286377, 8.642212289460703 ], "difference": 7.653797049885634, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__pythia-31m": { "ci": [ 9.944853555192541, 11.916576975073232 ], "difference": 10.919025791271682, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__slm-10m": { "ci": [ 6.635890360218745, 8.602630427540772 ], "difference": 7.620324312434939, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "gpt2-124m__supra-50m-base": { "ci": [ -0.22231016068905624, 1.4204925503246024 ], "difference": 0.5876261353332725, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "gpt2-124m__veyra2-apricot-50m-base": { "ci": [ 0.9211716297796128, 2.705342689004217 ], "difference": 1.8070719755705327, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__atom-3-4m": { "ci": [ 27.58893756755036, 30.042187485115143 ], "difference": 28.789996315485123, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__baguettotron": { "ci": [ 12.443170128806557, 14.63607099576427 ], "difference": 13.526740651870819, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__falcon-h1-tiny-r-90m": { "ci": [ 21.180647286515022, 23.545436928921546 ], "difference": 22.34695811824909, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__gemma-3-270m": { "ci": [ 3.2295084969038235, 5.224620472855046 ], "difference": 4.247314979070493, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__gpt-s-1-4m": { "ci": [ 29.546430185030548, 32.048984499243076 ], "difference": 30.822639445627008, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__gpt-s2-5m": { "ci": [ 22.692921220894892, 25.060730091228283 ], "difference": 23.858188709910568, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__gpt-x2-125m": { "ci": [ 6.543548887485459, 8.719161211825703 ], "difference": 7.643037384044087, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__gpt2-124m": { "ci": [ 14.143222563888733, 16.4118593800478 ], "difference": 15.266974006372491, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__mobilellm-r1-140m-base": { "ci": [ 10.873284350857205, 13.02640742452202 ], "difference": 11.955807929644676, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__nanowhale-100m-base": { "ci": [ 24.094314926003207, 26.46428926562647 ], "difference": 25.269477462631208, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__pythia-160m": { "ci": [ 21.76448313300386, 24.12912979103644 ], "difference": 22.920771056258122, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__pythia-31m": { "ci": [ 24.947883218281085, 27.411184938190967 ], "difference": 26.185999797644175, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__qwen2-5-0-5b": { "ci": [ -0.7937443045238336, 1.2109056554752804 ], "difference": 0.19496537145225148, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "lfm2-350m__slm-10m": { "ci": [ 21.736389241473272, 24.094793643932555 ], "difference": 22.887298318807428, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__smollm2-135m": { "ci": [ 4.2639180400644365, 6.304742227756106 ], "difference": 5.305004387140445, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__supra-50m-base": { "ci": [ 14.717188310254356, 16.936606168548764 ], "difference": 15.854600141705763, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "lfm2-350m__veyra2-apricot-50m-base": { "ci": [ 15.922087105664474, 18.21523002545234 ], "difference": 17.074045981943023, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__atom-3-4m": { "ci": [ 15.724389468592625, 17.9181576661227 ], "difference": 16.83418838584044, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__baguettotron": { "ci": [ 0.6028456289645157, 2.5682754519275246 ], "difference": 1.5709327222261427, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__falcon-h1-tiny-r-90m": { "ci": [ 9.366761121353298, 11.456567965416465 ], "difference": 10.391150188604414, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__gpt-s-1-4m": { "ci": [ 17.72035213429454, 19.952356120653356 ], "difference": 18.86683151598233, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__gpt-s2-5m": { "ci": [ 10.837860366434668, 12.916752286983545 ], "difference": 11.902380780265888, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__gpt2-124m": { "ci": [ 2.3527040000432247, 4.2604573197727165 ], "difference": 3.3111660767278153, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__nanowhale-100m-base": { "ci": [ 12.244547029232189, 14.383753052584806 ], "difference": 13.313669532986532, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__pythia-160m": { "ci": [ 9.850460315653072, 12.110903533239728 ], "difference": 10.96496312661345, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__pythia-31m": { "ci": [ 13.173488573335227, 15.260343800248487 ], "difference": 14.230191867999498, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__slm-10m": { "ci": [ 9.834008908382621, 11.962969758352465 ], "difference": 10.931490389162754, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__supra-50m-base": { "ci": [ 2.9471642992203297, 4.800180307345809 ], "difference": 3.898792212061088, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "mobilellm-r1-140m-base__veyra2-apricot-50m-base": { "ci": [ 4.204117362859041, 6.037751350789785 ], "difference": 5.118238052298348, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "nanowhale-100m-base__atom-3-4m": { "ci": [ 2.5114800270993958, 4.522962101886533 ], "difference": 3.520518852853912, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "nanowhale-100m-base__gpt-s-1-4m": { "ci": [ 4.520390817955864, 6.571451905864101 ], "difference": 5.553161982995798, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "nanowhale-100m-base__pythia-31m": { "ci": [ -0.04982526813426013, 1.8445162857873265 ], "difference": 0.9165223350129663, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "pythia-160m__atom-3-4m": { "ci": [ 4.737172915984907, 7.018566565651529 ], "difference": 5.8692252592269964, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "pythia-160m__gpt-s-1-4m": { "ci": [ 6.721804247693422, 9.058153968300607 ], "difference": 7.901868389368882, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "pythia-160m__gpt-s2-5m": { "ci": [ -0.19073668035679597, 2.031592137727357 ], "difference": 0.9374176536524387, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "pythia-160m__nanowhale-100m-base": { "ci": [ 1.24315052655882, 3.4148330794518693 ], "difference": 2.348706406373083, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "pythia-160m__pythia-31m": { "ci": [ 2.148599543800454, 4.357216741831833 ], "difference": 3.2652287413860495, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "pythia-31m__atom-3-4m": { "ci": [ 1.6057485508016338, 3.598754829970397 ], "difference": 2.6039965178409457, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "pythia-31m__gpt-s-1-4m": { "ci": [ 3.553404497040384, 5.681548302264185 ], "difference": 4.636639647982832, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__atom-3-4m": { "ci": [ 27.429065503800224, 29.758126578860814 ], "difference": 28.59503094403287, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__baguettotron": { "ci": [ 12.233474837251677, 14.410463289394922 ], "difference": 13.331775280418565, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__falcon-h1-tiny-r-90m": { "ci": [ 21.042544193912452, 23.292755794541097 ], "difference": 22.151992746796836, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__gemma-3-270m": { "ci": [ 3.1289293385506296, 4.949670349757808 ], "difference": 4.0523496076182415, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__gpt-s-1-4m": { "ci": [ 29.358591326483403, 31.787910953580926 ], "difference": 30.627674074174756, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__gpt-s2-5m": { "ci": [ 22.484367687263422, 24.80643089248962 ], "difference": 23.663223338458312, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__gpt-x2-125m": { "ci": [ 6.504515401914428, 8.401385534393647 ], "difference": 7.448072012591836, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__gpt2-124m": { "ci": [ 14.119663769992147, 16.08811450435093 ], "difference": 15.07200863492024, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__mobilellm-r1-140m-base": { "ci": [ 10.797843611952915, 12.719193410831675 ], "difference": 11.760842558192424, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__nanowhale-100m-base": { "ci": [ 23.99988085171323, 26.190262920767886 ], "difference": 25.074512091178956, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__pythia-160m": { "ci": [ 21.56004563074552, 23.87905437950645 ], "difference": 22.725805684805874, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__pythia-31m": { "ci": [ 24.841628881519235, 27.14392774955085 ], "difference": 25.991034426191924, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__slm-10m": { "ci": [ 21.573153354252433, 23.791472841951883 ], "difference": 22.69233294735518, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__smollm2-135m": { "ci": [ 4.182377294645956, 6.041752403489344 ], "difference": 5.110039015688194, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__supra-50m-base": { "ci": [ 14.674719371285, 16.64642100282309 ], "difference": 15.659634770253513, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "qwen2-5-0-5b__veyra2-apricot-50m-base": { "ci": [ 15.843063351497843, 17.88305005638959 ], "difference": 16.87908061049077, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "slm-10m__atom-3-4m": { "ci": [ 5.036334220813094, 6.855128264261082 ], "difference": 5.902697996677689, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "slm-10m__gpt-s-1-4m": { "ci": [ 6.926706396733146, 8.94055735587147 ], "difference": 7.935341126819576, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "slm-10m__gpt-s2-5m": { "ci": [ 0.06659882140217428, 1.8222234812887246 ], "difference": 0.970890391103133, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "slm-10m__nanowhale-100m-base": { "ci": [ 1.4914905145647108, 3.2592393440485696 ], "difference": 2.382179143823778, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "slm-10m__pythia-160m": { "ci": [ -1.0366806397870105, 1.1714324189900007 ], "difference": 0.03347273745069435, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "tie" }, "slm-10m__pythia-31m": { "ci": [ 2.372327463312847, 4.223426615282993 ], "difference": 3.2987014788367444, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__atom-3-4m": { "ci": [ 22.34756564971431, 24.615212198542363 ], "difference": 23.484991928344677, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__baguettotron": { "ci": [ 7.21981046049257, 9.250291115821636 ], "difference": 8.221736264730373, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__falcon-h1-tiny-r-90m": { "ci": [ 15.998703266315285, 18.097428607290865 ], "difference": 17.04195373110864, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__gpt-s-1-4m": { "ci": [ 24.306597878265215, 26.660924664410892 ], "difference": 25.517635058486558, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__gpt-s2-5m": { "ci": [ 17.474818949264733, 19.65927462835377 ], "difference": 18.553184322770118, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__gpt-x2-125m": { "ci": [ 1.520836731078651, 3.1770188989552963 ], "difference": 2.338032996903642, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__gpt2-124m": { "ci": [ 9.01471391018264, 10.912959371692486 ], "difference": 9.961969619232047, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__mobilellm-r1-140m-base": { "ci": [ 5.754466156999673, 7.575045374739086 ], "difference": 6.65080354250423, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__nanowhale-100m-base": { "ci": [ 18.89929834073857, 21.076546736195468 ], "difference": 19.964473075490766, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__pythia-160m": { "ci": [ 16.533556862312427, 18.762458693268172 ], "difference": 17.61576666911768, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__pythia-31m": { "ci": [ 19.743832355893147, 21.98849774424981 ], "difference": 20.88099541050373, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__slm-10m": { "ci": [ 16.48898449494742, 18.664225888028522 ], "difference": 17.58229393166699, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__supra-50m-base": { "ci": [ 9.632456329100759, 11.448445329599902 ], "difference": 10.549595754565319, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "smollm2-135m__veyra2-apricot-50m-base": { "ci": [ 10.828312999027693, 12.721441869861339 ], "difference": 11.769041594802578, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__atom-3-4m": { "ci": [ 11.943975471387882, 13.975141551589562 ], "difference": 12.935396173779356, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__falcon-h1-tiny-r-90m": { "ci": [ 5.4933752538658185, 7.494091690632844 ], "difference": 6.4923579765433255, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__gpt-s-1-4m": { "ci": [ 13.913111223226107, 16.026314781815554 ], "difference": 14.968039303921243, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__gpt-s2-5m": { "ci": [ 7.055711440473382, 8.973542914177806 ], "difference": 8.0035885682048, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__nanowhale-100m-base": { "ci": [ 8.481265325554629, 10.332046076912933 ], "difference": 9.414877320925445, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__pythia-160m": { "ci": [ 5.988387846530205, 8.160833376393306 ], "difference": 7.0661709145523615, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__pythia-31m": { "ci": [ 9.341259784482242, 11.317334294187644 ], "difference": 10.33139965593841, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__slm-10m": { "ci": [ 6.057935337460307, 8.006326905907825 ], "difference": 7.032698177101667, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "supra-50m-base__veyra2-apricot-50m-base": { "ci": [ 0.352626714684538, 2.1370426725718357 ], "difference": 1.21944584023726, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__atom-3-4m": { "ci": [ 10.627668222625106, 12.751269936412879 ], "difference": 11.715950333542096, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__falcon-h1-tiny-r-90m": { "ci": [ 4.269879876321943, 6.237640792156269 ], "difference": 5.272912136306066, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__gpt-s-1-4m": { "ci": [ 12.499782605339313, 14.838887566245907 ], "difference": 13.748593463683981, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__gpt-s2-5m": { "ci": [ 5.8032698087522885, 7.724554386452487 ], "difference": 6.784142727967539, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__nanowhale-100m-base": { "ci": [ 7.226025832871106, 9.190466152028234 ], "difference": 8.195431480688185, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__pythia-160m": { "ci": [ 4.834200955167154, 6.909557402281555 ], "difference": 5.846725074315101, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__pythia-31m": { "ci": [ 8.142263967111687, 10.097664917317006 ], "difference": 9.11195381570115, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" }, "veyra2-apricot-50m-base__slm-10m": { "ci": [ 4.863643976660482, 6.735987982048942 ], "difference": 5.813252336864407, "matched_samples": { "arc_challenge": 1172, "arc_easy": 2376, "blimp:Anaphor agreement": 2000, "blimp:Argument structure": 9000, "blimp:Binding": 7000, "blimp:Control / raising": 5000, "blimp:Determiner-noun agreement": 8000, "blimp:Ellipsis": 2000, "blimp:Filler-gap": 10000, "blimp:Irregular forms": 2000, "blimp:Island effects": 8000, "blimp:NPI licensing": 7000, "blimp:Quantifiers": 4000, "blimp:Subject-verb agreement": 6000, "commonsense_qa": 1221, "hellaswag": 10042, "piqa": 1838 }, "relation": "first_better" } }, "generated_at": "2026-07-22T07:57:07.788524+00:00", "models": [ { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/lfm2-350m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/lfm2-350m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.39590443686006827, "chance": 0.25, "ci": [ 15.696814562002272, 23.321956769055745 ], "classification": "reliably above chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 19.45392491467577 }, "arc_easy": { "accuracy": 0.6641414141414141, "chance": 0.25, "ci": [ 52.637485970819306, 57.744107744107744 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 55.21885521885522 }, "blimp": { "accuracy": 0.7906593253968254, "categories": { "Anaphor agreement": 0.977, "Argument structure": 0.7756666666666666, "Binding": 0.7418571428571428, "Control / raising": 0.8074, "Determiner-noun agreement": 0.9377500000000001, "Ellipsis": 0.8654999999999999, "Filler-gap": 0.7779999999999999, "Irregular forms": 0.728, "Island effects": 0.739, "NPI licensing": 0.5485714285714286, "Quantifiers": 0.7414999999999999, "Subject-verb agreement": 0.8476666666666666 }, "category_count": 12, "chance": 0.5, "ci": [ 57.22199222883596, 58.42621676587302 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 58.13186507936508 }, "commonsense_qa": { "accuracy": 0.45045045045045046, "chance": 0.2, "ci": [ 27.927927927927925, 34.88943488943489 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 31.306306306306304 }, "hellaswag": { "accuracy": 0.4898426608245369, "chance": 0.25, "ci": [ 30.61143198566022, 33.25433180641306 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 31.97902144327159 }, "piqa": { "accuracy": 0.6958650707290533, "chance": 0.5, "ci": [ 35.03808487486397, 43.2018498367791 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 39.17301414581067 } }, "complete": true, "composite": { "ci": [ 42.03623437890757, 44.21094708361269 ], "classification": "reliably above chance", "score": 43.17876224644195 }, "id": "lfm2-350m", "name": "LFM2 350M", "params_m": 354.484, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 354483968, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/lfm2-350m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/lfm2-350m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/lfm2-350m", "tokenizer": "models/lfm2-350m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 1, 2 ], "release_date": "2025-07-10", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/lfm2-350m/20260716T154647Z-v1.1", "source": "LiquidAI/LFM2-350M", "track": "instruction", "updated_at": "2026-07-16T15:46:57.444261+00:00", "vocab_size": 65536 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/qwen2-5-0-5b/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/qwen2-5-0-5b/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.3199658703071672, "chance": 0.25, "ci": [ 5.802047781569963, 13.083048919226394 ], "classification": "reliably above chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 9.328782707622297 }, "arc_easy": { "accuracy": 0.5845959595959596, "chance": 0.25, "ci": [ 41.86307519640854, 47.1955667789001 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 44.61279461279461 }, "blimp": { "accuracy": 0.8435170304232805, "categories": { "Anaphor agreement": 0.986, "Argument structure": 0.8395555555555556, "Binding": 0.7834285714285715, "Control / raising": 0.8469999999999999, "Determiner-noun agreement": 0.95575, "Ellipsis": 0.8895, "Filler-gap": 0.7902857142857144, "Irregular forms": 0.9515, "Island effects": 0.694125, "NPI licensing": 0.7531428571428572, "Quantifiers": 0.7372500000000001, "Subject-verb agreement": 0.8946666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 67.90119460978836, 68.90242162698412 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 68.70340608465611 }, "commonsense_qa": { "accuracy": 0.4095004095004095, "chance": 0.2, "ci": [ 22.91154791154791, 29.67086404586403 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 26.187551187551186 }, "hellaswag": { "accuracy": 0.5212109141605258, "chance": 0.25, "ci": [ 34.846976033990565, 37.50282148310429 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 36.161455221403436 }, "piqa": { "accuracy": 0.6942328618063112, "chance": 0.5, "ci": [ 34.71164309031556, 42.763873775843294 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 38.84657236126223 } }, "complete": true, "composite": { "ci": [ 41.86348233604306, 43.96668012300082 ], "classification": "reliably above chance", "score": 42.989729225325554 }, "id": "qwen2-5-0-5b", "name": "Qwen2.5 0.5B", "params_m": 494.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 494032768, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/qwen2-5-0-5b/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/qwen2-5-0-5b/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/qwen2.5-0.5b", "tokenizer": "models/qwen2.5-0.5b", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 1, 2 ], "release_date": "2024-09-15", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/qwen2-5-0-5b/20260716T154647Z-v1.1", "source": "Qwen/Qwen2.5-0.5B", "track": "base", "updated_at": "2026-07-16T15:47:02.275688+00:00", "vocab_size": 151936 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gemma-3-270m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gemma-3-270m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.27986348122866894, "chance": 0.25, "ci": [ 0.5688282138794095, 7.281001137656425 ], "classification": "reliably above chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 3.9817974971558585 }, "arc_easy": { "accuracy": 0.5728114478114478, "chance": 0.25, "ci": [ 40.404040404040394, 45.62289562289561 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 43.0415263748597 }, "blimp": { "accuracy": 0.8213405423280423, "categories": { "Anaphor agreement": 0.99, "Argument structure": 0.7935555555555555, "Binding": 0.7918571428571427, "Control / raising": 0.8374, "Determiner-noun agreement": 0.956, "Ellipsis": 0.87, "Filler-gap": 0.7927142857142858, "Irregular forms": 0.9295, "Island effects": 0.68875, "NPI licensing": 0.668142857142857, "Quantifiers": 0.7084999999999999, "Subject-verb agreement": 0.8296666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 63.49301306216931, 64.53917559523809 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 64.26810846560846 }, "commonsense_qa": { "accuracy": 0.4258804258804259, "chance": 0.2, "ci": [ 24.754299754299755, 31.71580671580671 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 28.235053235053236 }, "hellaswag": { "accuracy": 0.4144592710615415, "chance": 0.25, "ci": [ 20.65325632344155, 23.29582420500564 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 21.927902808205534 }, "piqa": { "accuracy": 0.6849836779107725, "chance": 0.5, "ci": [ 32.86180631120783, 41.240478781284004 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 36.99673558215451 } }, "complete": true, "composite": { "ci": [ 37.774028206553304, 39.9138118144519 ], "classification": "reliably above chance", "score": 38.912790926162764 }, "id": "gemma-3-270m", "name": "Gemma 3 270M", "params_m": 270.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "32", "batch_sizes": [], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 268098176, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gemma-3-270m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gemma-3-270m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/gemma-3-270m", "tokenizer": "models/gemma-3-270m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 3, 3 ], "release_date": "2025-08-05", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gemma-3-270m/20260716T154647Z-v1.1", "source": "google/gemma-3-270m", "track": "base", "updated_at": "2026-07-16T15:46:49.790812+00:00", "vocab_size": 262144 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/smollm2-135m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/smollm2-135m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.29692832764505117, "chance": 0.25, "ci": [ 2.6166097838452806, 9.670079635949946 ], "classification": "reliably above chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 6.2571103526734895 }, "arc_easy": { "accuracy": 0.5871212121212122, "chance": 0.25, "ci": [ 42.255892255892256, 47.53086419753087 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 44.949494949494955 }, "blimp": { "accuracy": 0.8136013227513228, "categories": { "Anaphor agreement": 0.977, "Argument structure": 0.8337777777777778, "Binding": 0.7632857142857142, "Control / raising": 0.8392000000000002, "Determiner-noun agreement": 0.968, "Ellipsis": 0.8745, "Filler-gap": 0.7977142857142857, "Irregular forms": 0.8380000000000001, "Island effects": 0.6892499999999999, "NPI licensing": 0.6485714285714286, "Quantifiers": 0.67025, "Subject-verb agreement": 0.8636666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 61.932142030423265, 63.0667154431217 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 62.72026455026456 }, "commonsense_qa": { "accuracy": 0.35544635544635544, "chance": 0.2, "ci": [ 16.152231777231773, 22.81173218673217 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 19.43079443079443 }, "hellaswag": { "accuracy": 0.43218482374029077, "chance": 0.25, "ci": [ 23.016331408086035, 25.592511451902016 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 24.291309832038767 }, "piqa": { "accuracy": 0.6817192600652884, "chance": 0.5, "ci": [ 32.097388465723604, 40.58759521218715 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 36.34385201305768 } }, "complete": true, "composite": { "ci": [ 36.68113506339092, 38.849860312348774 ], "classification": "reliably above chance", "score": 37.843491883620345 }, "id": "smollm2-135m", "name": "SmolLM2 135M", "params_m": 135.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 134515008, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/smollm2-135m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/smollm2-135m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/smollm2-135m", "tokenizer": "models/smollm2-135m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 4, 4 ], "release_date": "2024-10-31", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/smollm2-135m/20260716T154647Z-v1.1", "source": "HuggingFaceTB/SmolLM2-135M", "track": "base", "updated_at": "2026-07-16T15:47:05.346313+00:00", "vocab_size": 49152 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-x2-125m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-x2-125m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.27559726962457337, "chance": 0.25, "ci": [ -0.11376564277588337, 7.053469852104666 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 3.4129692832764498 }, "arc_easy": { "accuracy": 0.5164141414141414, "chance": 0.25, "ci": [ 32.82687991021324, 38.21548821548822 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 35.52188552188553 }, "blimp": { "accuracy": 0.8291201719576721, "categories": { "Anaphor agreement": 0.9864999999999999, "Argument structure": 0.826111111111111, "Binding": 0.7557142857142857, "Control / raising": 0.8231999999999999, "Determiner-noun agreement": 0.9581249999999999, "Ellipsis": 0.8534999999999999, "Filler-gap": 0.7958571428571429, "Irregular forms": 0.923, "Island effects": 0.6868749999999999, "NPI licensing": 0.7461428571428571, "Quantifiers": 0.70725, "Subject-verb agreement": 0.8871666666666665 }, "category_count": 12, "chance": 0.5, "ci": [ 64.96420122354498, 66.0613258928571 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 65.82403439153441 }, "commonsense_qa": { "accuracy": 0.3464373464373464, "chance": 0.2, "ci": [ 14.926289926289922, 21.580671580671574 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 18.3046683046683 }, "hellaswag": { "accuracy": 0.4051981676956781, "chance": 0.25, "ci": [ 19.431388169687313, 21.941512314943907 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 20.693089026090416 }, "piqa": { "accuracy": 0.6708378672470077, "chance": 0.5, "ci": [ 29.706202393906423, 38.41131664853101 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 34.16757344940153 } }, "complete": true, "composite": { "ci": [ 34.34048401085747, 36.481411360993164 ], "classification": "reliably above chance", "score": 35.52389451013529 }, "id": "gpt-x2-125m", "name": "GPT-X2 125M", "params_m": 125.081664, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 125081664, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-x2-125m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-x2-125m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-x2-125m", "tokenizer": "models/gpt-x2-125m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 5, 5 ], "release_date": "2026-03-25", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-x2-125m/20260716T154647Z-v1.1", "source": "AxiomicLabs/GPT-X2-125M", "track": "base", "updated_at": "2026-07-16T15:46:54.567291+00:00", "vocab_size": 32768 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/mobilellm-r1-140m-base/20260716T161918Z/models__mobilellm-r1-140m-base/results_2026-07-16T18-27-38.828116.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/mobilellm-r1-140m-base/20260716T161918Z" }, "benchmarks": { "arc_challenge": { "accuracy": 0.24146757679180889, "chance": 0.25, "ci": [ -4.550625711035269, 2.2753128555176305 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -1.1376564277588153 }, "arc_easy": { "accuracy": 0.49915824915824913, "chance": 0.25, "ci": [ 30.47138047138047, 35.746352413019075 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 33.22109988776655 }, "blimp": { "accuracy": 0.8078363425925926, "categories": { "Anaphor agreement": 0.9735, "Argument structure": 0.7977777777777777, "Binding": 0.7627142857142858, "Control / raising": 0.8108000000000001, "Determiner-noun agreement": 0.9315, "Ellipsis": 0.892, "Filler-gap": 0.767, "Irregular forms": 0.9495, "Island effects": 0.6956249999999999, "NPI licensing": 0.6222857142857142, "Quantifiers": 0.6245, "Subject-verb agreement": 0.8668333333333335 }, "category_count": 12, "chance": 0.5, "ci": [ 60.729346891534426, 61.79221445105819 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 61.567268518518524 }, "commonsense_qa": { "accuracy": 0.3546273546273546, "chance": 0.2, "ci": [ 15.950040950040947, 22.809172809172804 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 19.328419328419326 }, "hellaswag": { "accuracy": 0.3406691894045011, "chance": 0.25, "ci": [ 10.893580296089754, 13.297815840138082 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 12.089225253933478 }, "piqa": { "accuracy": 0.6327529923830251, "chance": 0.5, "ci": [ 22.19804134929271, 30.794341675734493 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 26.550598476605014 } }, "complete": true, "composite": { "ci": [ 30.06941936355877, 32.21641211384041 ], "classification": "reliably above chance", "score": 31.211043675010572 }, "id": "mobilellm-r1-140m-base", "name": "MobileLLM-R1 140M Base", "params_m": 140.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 140248512, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "torch_seed": 1234, "use_cache": null }, "n_shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa_text": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/mobilellm-r1-140m-base", "tokenizer": "models/mobilellm-r1-140m-base", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 6, 6 ], "release_date": "2025-09-10", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/mobilellm-r1-140m-base/20260716T161918Z", "source": "facebook/MobileLLM-R1-140M-base", "track": "base", "updated_at": "2026-07-16T16:27:44.727364+00:00", "vocab_size": 128256 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/models__baguettotron/results_2026-07-20T01-21-17.781285.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z" }, "benchmarks": { "arc_challenge": { "accuracy": 0.3037542662116041, "chance": 0.25, "ci": [ 3.7542662116040924, 10.58304891922638 ], "classification": "reliably above chance", "label": "ARC-Challenge", "n_samples": 1172, "score": 7.16723549488055 }, "arc_easy": { "accuracy": 0.5058922558922558, "chance": 0.25, "ci": [ 31.48148148148148, 36.700336700336706 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 34.11896745230078 }, "blimp": { "accuracy": 0.7847246362433862, "categories": { "Anaphor agreement": 0.8355, "Argument structure": 0.7777777777777778, "Binding": 0.7462857142857143, "Control / raising": 0.7584, "Determiner-noun agreement": 0.941, "Ellipsis": 0.88, "Filler-gap": 0.7588571428571429, "Irregular forms": 0.915, "Island effects": 0.630375, "NPI licensing": 0.623, "Quantifiers": 0.7064999999999999, "Subject-verb agreement": 0.8439999999999999 }, "category_count": 12, "chance": 0.5, "ci": [ 56.01578158068786, 57.203216600529124 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 56.94492724867723 }, "commonsense_qa": { "accuracy": 0.3063063063063063, "chance": 0.2, "ci": [ 10.012285012285009, 16.56429156429156 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 13.288288288288284 }, "hellaswag": { "accuracy": 0.35391356303525195, "chance": 0.25, "ci": [ 12.607050388368853, 15.050122817499838 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 13.855141738033595 }, "piqa": { "accuracy": 0.6207834602829162, "chance": 0.5, "ci": [ 19.583786724700754, 28.726877040261158 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 24.15669205658324 } }, "complete": true, "composite": { "ci": [ 28.428847552063168, 30.68468269023897 ], "classification": "reliably above chance", "score": 29.63151077505659 }, "id": "baguettotron", "name": "Baguettotron", "params_m": 321.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": "dc3c1cf", "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 320956992, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "torch_seed": 1234, "use_cache": null }, "n_shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa_text": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/baguettotron", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 7, 7 ], "release_date": "2025-11-10", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/baguettotron/20260719T231314Z", "source": "PleIAs/Baguettotron", "track": "instruction", "updated_at": "2026-07-19T23:21:23.658293+00:00", "vocab_size": 65536 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt2-124m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt2-124m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.22440273037542663, "chance": 0.25, "ci": [ -6.598407281001136, -0.22753128555176305 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -3.4129692832764498 }, "arc_easy": { "accuracy": 0.3947811447811448, "chance": 0.25, "ci": [ 16.665263748597084, 21.885521885521886 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 19.304152637485974 }, "blimp": { "accuracy": 0.8377077380952381, "categories": { "Anaphor agreement": 0.997, "Argument structure": 0.8356666666666667, "Binding": 0.79, "Control / raising": 0.8006, "Determiner-noun agreement": 0.953875, "Ellipsis": 0.848, "Filler-gap": 0.8117142857142856, "Irregular forms": 0.958, "Island effects": 0.693625, "NPI licensing": 0.7674285714285715, "Quantifiers": 0.7177500000000001, "Subject-verb agreement": 0.8788333333333332 }, "category_count": 12, "chance": 0.5, "ci": [ 66.76148528439154, 67.78422933201058 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 67.54154761904762 }, "commonsense_qa": { "accuracy": 0.29811629811629814, "chance": 0.2, "ci": [ 9.193284193284189, 15.54054054054054 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 12.264537264537264 }, "hellaswag": { "accuracy": 0.31139215295757816, "chance": 0.25, "ci": [ 7.003916882427141, 9.380601473809998 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 8.185620394343756 }, "piqa": { "accuracy": 0.6251360174102285, "chance": 0.5, "ci": [ 20.67464635473342, 29.37976060935801 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 25.0272034820457 } }, "complete": true, "composite": { "ci": [ 26.76217046436712, 28.856641117490366 ], "classification": "reliably above chance", "score": 27.871165134240197 }, "id": "gpt2-124m", "name": "GPT-2 124M", "params_m": 124.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false }, "model_dtype": "torch.float32", "model_num_parameters": 124439808, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt2-124m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt2-124m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "float32", "pretrained": "models/gpt2-124m", "tokenizer": "models/gpt2-124m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 8, 9 ], "release_date": "2022-03-02", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt2-124m/20260716T154647Z-v1.1", "source": "openai-community/gpt2", "track": "base", "updated_at": "2026-07-16T15:46:55.978943+00:00", "vocab_size": 50257 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.24914675767918087, "chance": 0.25, "ci": [ -3.526734926052333, 3.1854379977246903 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -0.11376564277588337 }, "arc_easy": { "accuracy": 0.4562289562289562, "chance": 0.25, "ci": [ 24.74747474747475, 30.1921997755331 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 27.497194163860826 }, "blimp": { "accuracy": 0.7947869047619047, "categories": { "Anaphor agreement": 0.913, "Argument structure": 0.7863333333333333, "Binding": 0.7452857142857142, "Control / raising": 0.8018000000000001, "Determiner-noun agreement": 0.9412499999999999, "Ellipsis": 0.857, "Filler-gap": 0.7838571428571429, "Irregular forms": 0.8845000000000001, "Island effects": 0.63425, "NPI licensing": 0.624, "Quantifiers": 0.6865, "Subject-verb agreement": 0.8796666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 57.94834143518519, 59.13693105158729 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 58.957380952380944 }, "commonsense_qa": { "accuracy": 0.29975429975429974, "chance": 0.2, "ci": [ 9.193284193284189, 15.74529074529074 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 12.469287469287464 }, "hellaswag": { "accuracy": 0.31627165903206533, "chance": 0.25, "ci": [ 7.6279625572595116, 9.991369581092746 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 8.836221204275377 }, "piqa": { "accuracy": 0.6169749727965179, "chance": 0.5, "ci": [ 18.933623503808494, 27.967899891186054 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 23.394994559303584 } }, "complete": true, "composite": { "ci": [ 26.15776710993005, 28.30631083392145 ], "classification": "reliably above chance", "score": 27.338042663290054 }, "id": "supra-50m-base", "name": "Supra 50M Base", "params_m": 51.78624, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 51786240, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T153917Z/tokenizer_compat", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T153917Z/tokenizer_compat", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/supra-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T094052Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 8, 9 ], "release_date": "2026-05-21", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/supra-50m-base/20260716T154647Z-v1.1", "source": "SupraLabs/Supra-50M-Base", "track": "base", "updated_at": "2026-07-16T15:47:07.077007+00:00", "vocab_size": 32000 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/models__veyra2-apricot-50m-base/results_2026-07-16T18-37-33.567087.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z" }, "benchmarks": { "arc_challenge": { "accuracy": 0.23464163822525597, "chance": 0.25, "ci": [ -5.233219567690558, 1.137656427758819 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -2.047781569965871 }, "arc_easy": { "accuracy": 0.4276094276094276, "chance": 0.25, "ci": [ 20.93153759820427, 26.430976430976433 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 23.681257014590347 }, "blimp": { "accuracy": 0.7935405423280423, "categories": { "Anaphor agreement": 0.9115, "Argument structure": 0.7822222222222223, "Binding": 0.7711428571428571, "Control / raising": 0.7988, "Determiner-noun agreement": 0.933125, "Ellipsis": 0.849, "Filler-gap": 0.7371428571428572, "Irregular forms": 0.907, "Island effects": 0.676375, "NPI licensing": 0.6524285714285715, "Quantifiers": 0.67225, "Subject-verb agreement": 0.8315 }, "category_count": 12, "chance": 0.5, "ci": [ 57.73771858465608, 58.93828703703703 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 58.708108465608454 }, "commonsense_qa": { "accuracy": 0.28665028665028663, "chance": 0.2, "ci": [ 7.657657657657656, 14.004914004914006 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 10.831285831285827 }, "hellaswag": { "accuracy": 0.3127862975502888, "chance": 0.25, "ci": [ 7.203080395671517, 9.540264223594237 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 8.371506340038508 }, "piqa": { "accuracy": 0.6169749727965179, "chance": 0.5, "ci": [ 19.0424374319913, 27.856365614798694 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 23.394994559303584 } }, "complete": true, "composite": { "ci": [ 24.919591817425154, 27.053619924780357 ], "classification": "reliably above chance", "score": 26.088907580286257 }, "id": "veyra2-apricot-50m-base", "name": "Veyra2 Apricot 50M Base", "params_m": 49.3, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "64", "batch_sizes": [], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 49303040, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "torch_seed": 1234, "use_cache": null }, "n_shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa_text": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/veyra2-apricot-50m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 10, 10 ], "release_date": "2026-06-14", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/veyra2-apricot-50m-base/20260716T163209Z", "source": "veyra-ai/Veyra2-Apricot-50M-Base", "track": "base", "updated_at": "2026-07-16T16:37:38.898399+00:00", "vocab_size": 8192 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/falcon-h1-tiny-r-90m/20260720T072155Z/models__falcon-h1-tiny-r-90m/results_2026-07-20T09-28-46.371565.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/falcon-h1-tiny-r-90m/20260720T072155Z" }, "benchmarks": { "arc_challenge": { "accuracy": 0.24914675767918087, "chance": 0.25, "ci": [ -3.415813424345847, 3.1854379977246903 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -0.11376564277588337 }, "arc_easy": { "accuracy": 0.3926767676767677, "chance": 0.25, "ci": [ 16.273849607182942, 21.49410774410773 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 19.023569023569024 }, "blimp": { "accuracy": 0.7126083333333333, "categories": { "Anaphor agreement": 0.8009999999999999, "Argument structure": 0.6073333333333334, "Binding": 0.7232857142857142, "Control / raising": 0.6587999999999999, "Determiner-noun agreement": 0.8799999999999999, "Ellipsis": 0.871, "Filler-gap": 0.7327142857142858, "Irregular forms": 0.781, "Island effects": 0.5905, "NPI licensing": 0.704, "Quantifiers": 0.44050000000000006, "Subject-verb agreement": 0.7611666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 41.63552711640213, 42.98391468253967 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 42.52166666666666 }, "commonsense_qa": { "accuracy": 0.2833742833742834, "chance": 0.2, "ci": [ 7.350532350532346, 13.595413595413595 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 10.42178542178542 }, "hellaswag": { "accuracy": 0.30770762796255724, "chance": 0.25, "ci": [ 6.552479585739893, 8.876053906924254 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 7.694350395007632 }, "piqa": { "accuracy": 0.6077257889009793, "chance": 0.5, "ci": [ 16.97497279651796, 25.897714907508163 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 21.545157780195858 } }, "complete": true, "composite": { "ci": [ 19.64362638928377, 21.818589193772866 ], "classification": "reliably above chance", "score": 20.800441105410517 }, "id": "falcon-h1-tiny-r-90m", "name": "Falcon H1 Tiny R 90M", "params_m": 90.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": "dc3c1cf", "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 91131072, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "torch_seed": 1234, "use_cache": null }, "n_shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa_text": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/falcon-h1-tiny-r-90m", "tokenizer": "models/falcon-h1-tiny-r-90m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 11, 13 ], "release_date": "2026-01-12", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/falcon-h1-tiny-r-90m/20260720T072155Z", "source": "tiiuae/Falcon-H1-Tiny-R-90M", "track": "instruction", "updated_at": "2026-07-20T07:28:52.389064+00:00", "vocab_size": 32768 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.2354948805460751, "chance": 0.25, "ci": [ -5.1222980659840704, 1.251422070534695 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -1.9340159271899877 }, "arc_easy": { "accuracy": 0.3569023569023569, "chance": 0.25, "ci": [ 11.560044893378228, 16.835016835016834 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 14.253647586980918 }, "blimp": { "accuracy": 0.772000496031746, "categories": { "Anaphor agreement": 0.8959999999999999, "Argument structure": 0.7603333333333333, "Binding": 0.7562857142857142, "Control / raising": 0.757, "Determiner-noun agreement": 0.89825, "Ellipsis": 0.7855, "Filler-gap": 0.7295714285714286, "Irregular forms": 0.9505, "Island effects": 0.508875, "NPI licensing": 0.7078571428571429, "Quantifiers": 0.6639999999999999, "Subject-verb agreement": 0.8498333333333333 }, "category_count": 12, "chance": 0.5, "ci": [ 53.51157457010582, 54.73998561507939 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 54.4000992063492 }, "commonsense_qa": { "accuracy": 0.2596232596232596, "chance": 0.2, "ci": [ 4.484029484029484, 10.62653562653563 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 7.452907452907449 }, "hellaswag": { "accuracy": 0.2731527584146584, "chance": 0.25, "ci": [ 1.958109274380932, 4.228905264555535 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 3.087034455287789 }, "piqa": { "accuracy": 0.5707290533188248, "chance": 0.5, "ci": [ 9.681719260065288, 18.50108813928181 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 14.14581066376497 } }, "complete": true, "composite": { "ci": [ 19.071533266309793, 21.225442487059897 ], "classification": "reliably above chance", "score": 20.26599982402355 }, "id": "slm-10m", "name": "SLM 10M", "params_m": 9.96864, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 9968640, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T153812Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T153812Z/tokenizer_compat", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/slm-10m", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T101325Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 11, 13 ], "release_date": "2026-06-13", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/slm-10m/20260716T154647Z-v1.1", "source": "liodon-ai/slm-10m", "track": "base", "updated_at": "2026-07-16T15:47:03.814706+00:00", "vocab_size": 8192 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-160m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-160m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.23037542662116042, "chance": 0.25, "ci": [ -5.688282138794084, 0.5688282138794095 ], "classification": "indistinguishable from chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -2.616609783845277 }, "arc_easy": { "accuracy": 0.3686868686868687, "chance": 0.25, "ci": [ 13.18742985409652, 18.350168350168353 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 15.824915824915822 }, "blimp": { "accuracy": 0.7286577050264551, "categories": { "Anaphor agreement": 0.9245000000000001, "Argument structure": 0.7542222222222222, "Binding": 0.7448571428571429, "Control / raising": 0.7462, "Determiner-noun agreement": 0.826125, "Ellipsis": 0.759, "Filler-gap": 0.6882857142857144, "Irregular forms": 0.723, "Island effects": 0.631, "NPI licensing": 0.5852857142857142, "Quantifiers": 0.63075, "Subject-verb agreement": 0.7306666666666667 }, "category_count": 12, "chance": 0.5, "ci": [ 44.78496957671956, 46.238928736772515 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 45.73154100529102 }, "commonsense_qa": { "accuracy": 0.2932022932022932, "chance": 0.2, "ci": [ 8.37428337428337, 14.823914823914825 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 11.650286650286647 }, "hellaswag": { "accuracy": 0.30501892053375823, "chance": 0.25, "ci": [ 6.167430126800774, 8.504282015534756 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 7.335856071167764 }, "piqa": { "accuracy": 0.5843307943416758, "chance": 0.5, "ci": [ 12.51360174102285, 21.218715995647443 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 16.866158868335155 } }, "complete": true, "composite": { "ci": [ 19.111437730092515, 21.23647682872686 ], "classification": "reliably above chance", "score": 20.2119024553024 }, "id": "pythia-160m", "name": "Pythia 160M", "params_m": 160.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 162322944, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-160m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-160m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-160m", "tokenizer": "models/pythia-160m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 11, 14 ], "release_date": "2023-02-08", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-160m/20260716T154647Z-v1.1", "source": "EleutherAI/pythia-160m", "track": "base", "updated_at": "2026-07-16T15:46:58.993182+00:00", "vocab_size": 50304 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s2-5m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s2-5m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.22184300341296928, "chance": 0.25, "ci": [ -6.939704209328782, -0.6825938566552892 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -3.754266211604096 }, "arc_easy": { "accuracy": 0.335016835016835, "chance": 0.25, "ci": [ 8.585858585858587, 13.916947250280586 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 11.335578002244665 }, "blimp": { "accuracy": 0.7732386243386244, "categories": { "Anaphor agreement": 0.9405, "Argument structure": 0.7917777777777779, "Binding": 0.732857142857143, "Control / raising": 0.7488, "Determiner-noun agreement": 0.902375, "Ellipsis": 0.7815000000000001, "Filler-gap": 0.7285714285714285, "Irregular forms": 0.915, "Island effects": 0.538875, "NPI licensing": 0.7108571428571429, "Quantifiers": 0.63225, "Subject-verb agreement": 0.8555000000000001 }, "category_count": 12, "chance": 0.5, "ci": [ 53.86334275793652, 55.039749007936514 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 54.64772486772489 }, "commonsense_qa": { "accuracy": 0.24815724815724816, "chance": 0.2, "ci": [ 3.1505937755937747, 9.193284193284189 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 6.019656019656018 }, "hellaswag": { "accuracy": 0.27544313881696875, "chance": 0.25, "ci": [ 2.2638252672110504, 4.5083980614751376 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 3.392418508929166 }, "piqa": { "accuracy": 0.5647442872687704, "chance": 0.5, "ci": [ 8.26985854189337, 17.41022850924918 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 12.94885745375407 } }, "complete": true, "composite": { "ci": [ 18.15426094733613, 20.273369388072773 ], "classification": "reliably above chance", "score": 19.279532451183385 }, "id": "gpt-s2-5m", "name": "GPT-S2 5M", "params_m": 5.384, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 5384258, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s2-5m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s2-5m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s2-5m", "tokenizer": "models/gpt-s2-5m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 13, 14 ], "release_date": "2026-06-19", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s2-5m/20260716T154647Z-v1.1", "source": "AxiomicLabs/GPT-S2-5M", "track": "base", "updated_at": "2026-07-16T15:46:53.057636+00:00", "vocab_size": 4096 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/models__nanowhale-100m-base/results_2026-07-20T01-41-02.427304.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z" }, "benchmarks": { "arc_challenge": { "accuracy": 0.22440273037542663, "chance": 0.25, "ci": [ -6.484641638225256, -0.22753128555176305 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -3.4129692832764498 }, "arc_easy": { "accuracy": 0.3728956228956229, "chance": 0.25, "ci": [ 13.692480359147021, 19.023569023569024 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 16.38608305274972 }, "blimp": { "accuracy": 0.7091933531746032, "categories": { "Anaphor agreement": 0.794, "Argument structure": 0.7473333333333333, "Binding": 0.7195714285714286, "Control / raising": 0.6965999999999999, "Determiner-noun agreement": 0.8567499999999999, "Ellipsis": 0.7050000000000001, "Filler-gap": 0.6524285714285715, "Irregular forms": 0.917, "Island effects": 0.495625, "NPI licensing": 0.5134285714285715, "Quantifiers": 0.64475, "Subject-verb agreement": 0.7678333333333334 }, "category_count": 12, "chance": 0.5, "ci": [ 40.99927827380954, 42.33725859788362 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 41.83867063492064 }, "commonsense_qa": { "accuracy": 0.27764127764127766, "chance": 0.2, "ci": [ 6.63390663390663, 12.878787878787877 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 9.705159705159705 }, "hellaswag": { "accuracy": 0.27126070503883687, "chance": 0.25, "ci": [ 1.63977959237868, 3.976631481112662 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 2.8347606718449163 }, "piqa": { "accuracy": 0.573449401523395, "chance": 0.5, "ci": [ 10.228509249183904, 19.26006528835691 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 14.689880304678994 } }, "complete": true, "composite": { "ci": [ 16.768369462355462, 18.88831197015816 ], "classification": "reliably above chance", "score": 17.857047764529888 }, "id": "nanowhale-100m-base", "name": "nanowhale 100M Base", "params_m": 110.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": "dc3c1cf", "inference_config": { "batch_size": "32", "batch_sizes": [], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true }, "model_dtype": "torch.float32", "model_num_parameters": 110369621, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "torch_seed": 1234, "use_cache": null }, "n_shot": { "arc_challenge": 0, "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa_text": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "float32", "pretrained": "models/nanowhale-100m-base", "tokenizer": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z/tokenizer_compat", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 15, 16 ], "release_date": "2026-04-24", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/nanowhale-100m-base/20260719T232857Z", "source": "HuggingFaceTB/nanowhale-100m-base", "track": "base", "updated_at": "2026-07-19T23:41:08.863409+00:00", "vocab_size": 129280 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-31m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-31m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.2167235494880546, "chance": 0.25, "ci": [ -7.622298065984071, -1.1376564277588153 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -4.436860068259385 }, "arc_easy": { "accuracy": 0.3409090909090909, "chance": 0.25, "ci": [ 9.652076318742987, 14.534231200897866 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 12.121212121212118 }, "blimp": { "accuracy": 0.7167433531746031, "categories": { "Anaphor agreement": 0.9285, "Argument structure": 0.7563333333333333, "Binding": 0.7057142857142856, "Control / raising": 0.7332, "Determiner-noun agreement": 0.857375, "Ellipsis": 0.6745, "Filler-gap": 0.6648571428571428, "Irregular forms": 0.8815, "Island effects": 0.535, "NPI licensing": 0.45985714285714285, "Quantifiers": 0.62375, "Subject-verb agreement": 0.7803333333333334 }, "category_count": 12, "chance": 0.5, "ci": [ 42.51403703703704, 43.85099421296294 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 43.34867063492063 }, "commonsense_qa": { "accuracy": 0.26453726453726456, "chance": 0.2, "ci": [ 5.098280098280095, 11.138411138411136 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 8.067158067158068 }, "hellaswag": { "accuracy": 0.27185819557857, "chance": 0.25, "ci": [ 1.7725552678749275, 4.069906393148773 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 2.9144260771426658 }, "piqa": { "accuracy": 0.5680087051142546, "chance": 0.5, "ci": [ 9.140369967355833, 18.063112078346034 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 13.601741022850922 } }, "complete": true, "composite": { "ci": [ 15.828302485043055, 17.965860323148288 ], "classification": "reliably above chance", "score": 16.92981203838316 }, "id": "pythia-31m", "name": "Pythia 31M", "params_m": 31.0, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false }, "model_dtype": "torch.bfloat16", "model_num_parameters": 30494720, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-31m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-31m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/pythia-31m", "tokenizer": "models/pythia-31m", "trust_remote_code": false, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 15, 16 ], "release_date": "2026-02-24", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/pythia-31m/20260716T154647Z-v1.1", "source": "EleutherAI/pythia-31m", "track": "base", "updated_at": "2026-07-16T15:47:00.508307+00:00", "vocab_size": 50304 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/atom-3-4m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/atom-3-4m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.21843003412969283, "chance": 0.25, "ci": [ -7.394766780432309, -1.1376564277588153 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -4.209328782707622 }, "arc_easy": { "accuracy": 0.33080808080808083, "chance": 0.25, "ci": [ 8.361391694725029, 13.24354657687991 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 10.774410774410777 }, "blimp": { "accuracy": 0.6819768518518519, "categories": { "Anaphor agreement": 0.6659999999999999, "Argument structure": 0.7285555555555556, "Binding": 0.713857142857143, "Control / raising": 0.6869999999999999, "Determiner-noun agreement": 0.842125, "Ellipsis": 0.708, "Filler-gap": 0.6594285714285714, "Irregular forms": 0.845, "Island effects": 0.5431250000000001, "NPI licensing": 0.5187142857142858, "Quantifiers": 0.53975, "Subject-verb agreement": 0.7321666666666666 }, "category_count": 12, "chance": 0.5, "ci": [ 35.579373181216944, 36.99308052248677 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 36.39537037037037 }, "commonsense_qa": { "accuracy": 0.24733824733824733, "chance": 0.2, "ci": [ 2.9484029484029453, 9.193284193284189 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 5.917280917280914 }, "hellaswag": { "accuracy": 0.27614021111332404, "chance": 0.25, "ci": [ 2.356768240058423, 4.547566885746535 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 3.4853614817765384 }, "piqa": { "accuracy": 0.5571273122959739, "chance": 0.5, "ci": [ 6.855277475516863, 15.88683351468989 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 11.42546245919478 } }, "complete": true, "composite": { "ci": [ 13.263758832507774, 15.316045445100848 ], "classification": "reliably above chance", "score": 14.327312373584093 }, "id": "atom-3-4m", "name": "Atom 3.4M", "params_m": 3.4128, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true }, "model_dtype": "torch.float32", "model_num_parameters": 3412800, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/atom-3-4m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/atom-3-4m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/atom-3-4m", "tokenizer": "models/atom-3-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 17, 17 ], "release_date": "2026-06-19", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/atom-3-4m/20260716T154647Z-v1.1", "source": "UniversalComputingResearch/Atom3.4m", "track": "base", "updated_at": "2026-07-16T15:46:48.573093+00:00", "vocab_size": 4096 }, { "artifacts": { "result_file": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s-1-4m/20260716T154647Z-v1.1/results_v1.1_merged.json", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s-1-4m/20260716T154647Z-v1.1" }, "benchmarks": { "arc_challenge": { "accuracy": 0.22013651877133106, "chance": 0.25, "ci": [ -7.053469852104666, -0.6825938566552892 ], "classification": "below chance", "label": "ARC-Challenge", "n_samples": 1172, "score": -3.9817974971558585 }, "arc_easy": { "accuracy": 0.31776094276094274, "chance": 0.25, "ci": [ 6.509539842873177, 11.503928170594838 ], "classification": "reliably above chance", "label": "ARC-Easy", "n_samples": 2376, "score": 9.034792368125698 }, "blimp": { "accuracy": 0.6599056216931216, "categories": { "Anaphor agreement": 0.6275, "Argument structure": 0.6435555555555555, "Binding": 0.7425714285714285, "Control / raising": 0.6728000000000001, "Determiner-noun agreement": 0.779625, "Ellipsis": 0.6525000000000001, "Filler-gap": 0.6531428571428571, "Irregular forms": 0.8235, "Island effects": 0.5603750000000001, "NPI licensing": 0.41571428571428576, "Quantifiers": 0.6217499999999999, "Subject-verb agreement": 0.7258333333333332 }, "category_count": 12, "chance": 0.5, "ci": [ 31.20117526455027, 32.64844031084656 ], "classification": "reliably above chance", "label": "BLiMP", "n_samples": 70000, "score": 31.981124338624323 }, "commonsense_qa": { "accuracy": 0.23423423423423423, "chance": 0.2, "ci": [ 1.5125921375921367, 7.35309172809171 ], "classification": "reliably above chance", "label": "CommonsenseQA", "n_samples": 1221, "score": 4.279279279279277 }, "hellaswag": { "accuracy": 0.26847241585341564, "chance": 0.25, "ci": [ 1.3075084644493165, 3.5919139613622804 ], "classification": "reliably above chance", "label": "HellaSwag", "n_samples": 10042, "score": 2.4629887804554182 }, "piqa": { "accuracy": 0.5516866158868335, "chance": 0.5, "ci": [ 5.658324265505987, 14.692600652883549 ], "classification": "reliably above chance", "label": "PIQA", "n_samples": 1838, "score": 10.33732317736671 } }, "complete": true, "composite": { "ci": [ 11.210655224075424, 13.297313646044007 ], "classification": "reliably above chance", "score": 12.287040849624308 }, "id": "gpt-s-1-4m", "name": "GPT-S 1.4M", "params_m": 1.426, "protocol": { "bootstrap_replicates": 2000, "bootstrap_seed": 20260716, "harness": "lm-eval==0.4.12", "harness_git_hash": null, "inference_config": { "batch_size": "auto", "batch_sizes": [ 64 ], "bootstrap_iters": 100000, "device": "cuda:0", "fewshot_seed": 1234, "gen_kwargs": {}, "limit": null, "model": "hf", "model_args": { "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true }, "model_dtype": "torch.bfloat16", "model_num_parameters": 1426176, "model_revision": "main", "model_sha": "", "numpy_seed": 1234, "random_seed": 1234, "tasks": [ "hellaswag", "piqa", "arc_easy", "arc_challenge", "commonsense_qa_text", "blimp" ], "torch_seed": 1234, "use_cache": null, "v1_1_sources": { "rechecked_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s-1-4m/20260716T154341Z-v1.1/results_v1.1_merged.json", "unchanged_tasks_result": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s-1-4m/20260716T154341Z-v1.1/results_v1.1_merged.json" } }, "n_shot": { "arc_easy": 0, "blimp": 0, "blimp_adjunct_island": 0, "blimp_anaphor_gender_agreement": 0, "blimp_anaphor_number_agreement": 0, "blimp_animate_subject_passive": 0, "blimp_animate_subject_trans": 0, "blimp_causative": 0, "blimp_complex_NP_island": 0, "blimp_coordinate_structure_constraint_complex_left_branch": 0, "blimp_coordinate_structure_constraint_object_extraction": 0, "blimp_determiner_noun_agreement_1": 0, "blimp_determiner_noun_agreement_2": 0, "blimp_determiner_noun_agreement_irregular_1": 0, "blimp_determiner_noun_agreement_irregular_2": 0, "blimp_determiner_noun_agreement_with_adj_2": 0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 0, "blimp_determiner_noun_agreement_with_adjective_1": 0, "blimp_distractor_agreement_relational_noun": 0, "blimp_distractor_agreement_relative_clause": 0, "blimp_drop_argument": 0, "blimp_ellipsis_n_bar_1": 0, "blimp_ellipsis_n_bar_2": 0, "blimp_existential_there_object_raising": 0, "blimp_existential_there_quantifiers_1": 0, "blimp_existential_there_quantifiers_2": 0, "blimp_existential_there_subject_raising": 0, "blimp_expletive_it_object_raising": 0, "blimp_inchoative": 0, "blimp_intransitive": 0, "blimp_irregular_past_participle_adjectives": 0, "blimp_irregular_past_participle_verbs": 0, "blimp_irregular_plural_subject_verb_agreement_1": 0, "blimp_irregular_plural_subject_verb_agreement_2": 0, "blimp_left_branch_island_echo_question": 0, "blimp_left_branch_island_simple_question": 0, "blimp_matrix_question_npi_licensor_present": 0, "blimp_npi_present_1": 0, "blimp_npi_present_2": 0, "blimp_only_npi_licensor_present": 0, "blimp_only_npi_scope": 0, "blimp_passive_1": 0, "blimp_passive_2": 0, "blimp_principle_A_c_command": 0, "blimp_principle_A_case_1": 0, "blimp_principle_A_case_2": 0, "blimp_principle_A_domain_1": 0, "blimp_principle_A_domain_2": 0, "blimp_principle_A_domain_3": 0, "blimp_principle_A_reconstruction": 0, "blimp_regular_plural_subject_verb_agreement_1": 0, "blimp_regular_plural_subject_verb_agreement_2": 0, "blimp_sentential_negation_npi_licensor_present": 0, "blimp_sentential_negation_npi_scope": 0, "blimp_sentential_subject_island": 0, "blimp_superlative_quantifiers_1": 0, "blimp_superlative_quantifiers_2": 0, "blimp_tough_vs_raising_1": 0, "blimp_tough_vs_raising_2": 0, "blimp_transitive": 0, "blimp_wh_island": 0, "blimp_wh_questions_object_gap": 0, "blimp_wh_questions_subject_gap": 0, "blimp_wh_questions_subject_gap_long_distance": 0, "blimp_wh_vs_that_no_gap": 0, "blimp_wh_vs_that_no_gap_long_distance": 0, "blimp_wh_vs_that_with_gap": 0, "blimp_wh_vs_that_with_gap_long_distance": 0, "commonsense_qa": 0, "hellaswag": 0, "piqa": 0 }, "suite": "slm-benchmark-v1.1", "task_config": { "arc_challenge": { "dataset_name": "ARC-Challenge", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_challenge.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_challenge", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "arc_easy": { "dataset_name": "ARC-Easy", "dataset_path": "allenai/ai2_arc", "description": "", "doc_to_choice": "{{choices.text}}", "doc_to_decontamination_query": "Question: {{question}}\nAnswer:", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{choices.text}}", "doc_to_target": "{{choices.label.index(answerKey)}}", "doc_to_text": "Question: {{question}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/arc/arc_easy.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "arc_easy", "test_split": "test", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "blimp_adjunct_island": { "dataset_name": "adjunct_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/adjunct_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_adjunct_island", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_gender_agreement": { "dataset_name": "anaphor_gender_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_gender_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_gender_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_anaphor_number_agreement": { "dataset_name": "anaphor_number_agreement", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/anaphor_number_agreement.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_anaphor_number_agreement", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_passive": { "dataset_name": "animate_subject_passive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_passive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_passive", "unsafe_code": false, "validation_split": "train" }, "blimp_animate_subject_trans": { "dataset_name": "animate_subject_trans", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/animate_subject_trans.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_animate_subject_trans", "unsafe_code": false, "validation_split": "train" }, "blimp_causative": { "dataset_name": "causative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/causative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_causative", "unsafe_code": false, "validation_split": "train" }, "blimp_complex_NP_island": { "dataset_name": "complex_NP_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/complex_NP_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_complex_NP_island", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_complex_left_branch": { "dataset_name": "coordinate_structure_constraint_complex_left_branch", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_complex_left_branch.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_complex_left_branch", "unsafe_code": false, "validation_split": "train" }, "blimp_coordinate_structure_constraint_object_extraction": { "dataset_name": "coordinate_structure_constraint_object_extraction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/coordinate_structure_constraint_object_extraction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_coordinate_structure_constraint_object_extraction", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_1": { "dataset_name": "determiner_noun_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_2": { "dataset_name": "determiner_noun_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_1": { "dataset_name": "determiner_noun_agreement_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_irregular_2": { "dataset_name": "determiner_noun_agreement_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_2": { "dataset_name": "determiner_noun_agreement_with_adj_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_1": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_1", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adj_irregular_2": { "dataset_name": "determiner_noun_agreement_with_adj_irregular_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adj_irregular_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adj_irregular_2", "unsafe_code": false, "validation_split": "train" }, "blimp_determiner_noun_agreement_with_adjective_1": { "dataset_name": "determiner_noun_agreement_with_adjective_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/determiner_noun_agreement_with_adjective_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_determiner_noun_agreement_with_adjective_1", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relational_noun": { "dataset_name": "distractor_agreement_relational_noun", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relational_noun.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relational_noun", "unsafe_code": false, "validation_split": "train" }, "blimp_distractor_agreement_relative_clause": { "dataset_name": "distractor_agreement_relative_clause", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/distractor_agreement_relative_clause.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_distractor_agreement_relative_clause", "unsafe_code": false, "validation_split": "train" }, "blimp_drop_argument": { "dataset_name": "drop_argument", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/drop_argument.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_drop_argument", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_1": { "dataset_name": "ellipsis_n_bar_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_1", "unsafe_code": false, "validation_split": "train" }, "blimp_ellipsis_n_bar_2": { "dataset_name": "ellipsis_n_bar_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/ellipsis_n_bar_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_ellipsis_n_bar_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_object_raising": { "dataset_name": "existential_there_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_1": { "dataset_name": "existential_there_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_quantifiers_2": { "dataset_name": "existential_there_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_existential_there_subject_raising": { "dataset_name": "existential_there_subject_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/existential_there_subject_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_existential_there_subject_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_expletive_it_object_raising": { "dataset_name": "expletive_it_object_raising", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/expletive_it_object_raising.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_expletive_it_object_raising", "unsafe_code": false, "validation_split": "train" }, "blimp_inchoative": { "dataset_name": "inchoative", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/inchoative.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_inchoative", "unsafe_code": false, "validation_split": "train" }, "blimp_intransitive": { "dataset_name": "intransitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/intransitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_intransitive", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_adjectives": { "dataset_name": "irregular_past_participle_adjectives", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_adjectives.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_adjectives", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_past_participle_verbs": { "dataset_name": "irregular_past_participle_verbs", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_past_participle_verbs.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_past_participle_verbs", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_1": { "dataset_name": "irregular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_irregular_plural_subject_verb_agreement_2": { "dataset_name": "irregular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/irregular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_irregular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_echo_question": { "dataset_name": "left_branch_island_echo_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_echo_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_echo_question", "unsafe_code": false, "validation_split": "train" }, "blimp_left_branch_island_simple_question": { "dataset_name": "left_branch_island_simple_question", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/left_branch_island_simple_question.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_left_branch_island_simple_question", "unsafe_code": false, "validation_split": "train" }, "blimp_matrix_question_npi_licensor_present": { "dataset_name": "matrix_question_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/matrix_question_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_matrix_question_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_1": { "dataset_name": "npi_present_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_1", "unsafe_code": false, "validation_split": "train" }, "blimp_npi_present_2": { "dataset_name": "npi_present_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/npi_present_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_npi_present_2", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_licensor_present": { "dataset_name": "only_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_only_npi_scope": { "dataset_name": "only_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/only_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_only_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_1": { "dataset_name": "passive_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_1", "unsafe_code": false, "validation_split": "train" }, "blimp_passive_2": { "dataset_name": "passive_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/passive_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_passive_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_c_command": { "dataset_name": "principle_A_c_command", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_c_command.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_c_command", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_1": { "dataset_name": "principle_A_case_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_case_2": { "dataset_name": "principle_A_case_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_case_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_case_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_1": { "dataset_name": "principle_A_domain_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_1", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_2": { "dataset_name": "principle_A_domain_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_2", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_domain_3": { "dataset_name": "principle_A_domain_3", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_domain_3.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_domain_3", "unsafe_code": false, "validation_split": "train" }, "blimp_principle_A_reconstruction": { "dataset_name": "principle_A_reconstruction", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/principle_A_reconstruction.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_principle_A_reconstruction", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_1": { "dataset_name": "regular_plural_subject_verb_agreement_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_1", "unsafe_code": false, "validation_split": "train" }, "blimp_regular_plural_subject_verb_agreement_2": { "dataset_name": "regular_plural_subject_verb_agreement_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/regular_plural_subject_verb_agreement_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_regular_plural_subject_verb_agreement_2", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_licensor_present": { "dataset_name": "sentential_negation_npi_licensor_present", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_licensor_present.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_licensor_present", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_negation_npi_scope": { "dataset_name": "sentential_negation_npi_scope", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_negation_npi_scope.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_negation_npi_scope", "unsafe_code": false, "validation_split": "train" }, "blimp_sentential_subject_island": { "dataset_name": "sentential_subject_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/sentential_subject_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_sentential_subject_island", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_1": { "dataset_name": "superlative_quantifiers_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_1", "unsafe_code": false, "validation_split": "train" }, "blimp_superlative_quantifiers_2": { "dataset_name": "superlative_quantifiers_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/superlative_quantifiers_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_superlative_quantifiers_2", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_1": { "dataset_name": "tough_vs_raising_1", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_1.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_1", "unsafe_code": false, "validation_split": "train" }, "blimp_tough_vs_raising_2": { "dataset_name": "tough_vs_raising_2", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/tough_vs_raising_2.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_tough_vs_raising_2", "unsafe_code": false, "validation_split": "train" }, "blimp_transitive": { "dataset_name": "transitive", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/transitive.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_transitive", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_island": { "dataset_name": "wh_island", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_island.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_island", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_object_gap": { "dataset_name": "wh_questions_object_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_object_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_object_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap": { "dataset_name": "wh_questions_subject_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_questions_subject_gap_long_distance": { "dataset_name": "wh_questions_subject_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_questions_subject_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_questions_subject_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap": { "dataset_name": "wh_vs_that_no_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_no_gap_long_distance": { "dataset_name": "wh_vs_that_no_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_no_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_no_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap": { "dataset_name": "wh_vs_that_with_gap", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap", "unsafe_code": false, "validation_split": "train" }, "blimp_wh_vs_that_with_gap_long_distance": { "dataset_name": "wh_vs_that_with_gap_long_distance", "dataset_path": "nyu-mll/blimp", "description": "", "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_decontamination_query": "{{sentence_good}} {{sentence_bad}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_config": { "doc_to_choice": "{{[sentence_good, sentence_bad]}}", "doc_to_target": 0, "doc_to_text": "", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/blimp/wh_vs_that_with_gap_long_distance.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "blimp_wh_vs_that_with_gap_long_distance", "unsafe_code": false, "validation_split": "train" }, "commonsense_qa_text": { "dataset_path": "tau/commonsense_qa", "description": "", "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{ choices.text }}", "doc_to_target": "{{ choices.label.index(answerKey) }}", "doc_to_text": "Question: {{ question.strip() }}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/config/tasks/commonsense_qa_text.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "commonsense_qa_text", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "hellaswag": { "dataset_path": "Rowan/hellaswag", "description": "", "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_config": { "doc_to_choice": "choices", "doc_to_target": "{{label}}", "doc_to_text": "{{query}}", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": "", "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n", "repeats": 1, "should_decontaminate": false, "target_delimiter": " ", "task": "hellaswag", "training_split": "train", "unsafe_code": false, "validation_split": "validation" }, "piqa": { "dataset_path": "baber/piqa", "description": "", "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_decontamination_query": "goal", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_config": { "doc_to_choice": "{{[sol1, sol2]}}", "doc_to_target": "label", "doc_to_text": "Question: {{goal}}\nAnswer:", "fewshot_delimiter": "\n\n", "fewshot_indices": null, "gen_prefix": null, "process_docs": null, "sampler": "default", "samples": null, "split": null, "target_delimiter": " " }, "fewshot_delimiter": "\n\n", "metadata": { "config_source": "/home/max/PROJECTS/PythonProjects/mcpterm/venv/lib/python3.11/site-packages/lm_eval/tasks/piqa/piqa.yaml", "dtype": "bfloat16", "pretrained": "models/gpt-s-1-4m", "tokenizer": "models/gpt-s-1-4m", "trust_remote_code": true, "version": 1.0 }, "metric_list": [ { "aggregation": "mean", "higher_is_better": true, "metric": "acc" }, { "aggregation": "mean", "higher_is_better": true, "metric": "acc_norm" } ], "num_fewshot": 0, "output_type": "multiple_choice", "repeats": 1, "should_decontaminate": true, "target_delimiter": " ", "task": "piqa", "training_split": "train", "unsafe_code": false, "validation_split": "validation" } }, "task_versions": { "arc_challenge": 1.0, "arc_easy": 1.0, "blimp": "2.0", "blimp_adjunct_island": 1.0, "blimp_anaphor_gender_agreement": 1.0, "blimp_anaphor_number_agreement": 1.0, "blimp_animate_subject_passive": 1.0, "blimp_animate_subject_trans": 1.0, "blimp_causative": 1.0, "blimp_complex_NP_island": 1.0, "blimp_coordinate_structure_constraint_complex_left_branch": 1.0, "blimp_coordinate_structure_constraint_object_extraction": 1.0, "blimp_determiner_noun_agreement_1": 1.0, "blimp_determiner_noun_agreement_2": 1.0, "blimp_determiner_noun_agreement_irregular_1": 1.0, "blimp_determiner_noun_agreement_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adj_2": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_1": 1.0, "blimp_determiner_noun_agreement_with_adj_irregular_2": 1.0, "blimp_determiner_noun_agreement_with_adjective_1": 1.0, "blimp_distractor_agreement_relational_noun": 1.0, "blimp_distractor_agreement_relative_clause": 1.0, "blimp_drop_argument": 1.0, "blimp_ellipsis_n_bar_1": 1.0, "blimp_ellipsis_n_bar_2": 1.0, "blimp_existential_there_object_raising": 1.0, "blimp_existential_there_quantifiers_1": 1.0, "blimp_existential_there_quantifiers_2": 1.0, "blimp_existential_there_subject_raising": 1.0, "blimp_expletive_it_object_raising": 1.0, "blimp_inchoative": 1.0, "blimp_intransitive": 1.0, "blimp_irregular_past_participle_adjectives": 1.0, "blimp_irregular_past_participle_verbs": 1.0, "blimp_irregular_plural_subject_verb_agreement_1": 1.0, "blimp_irregular_plural_subject_verb_agreement_2": 1.0, "blimp_left_branch_island_echo_question": 1.0, "blimp_left_branch_island_simple_question": 1.0, "blimp_matrix_question_npi_licensor_present": 1.0, "blimp_npi_present_1": 1.0, "blimp_npi_present_2": 1.0, "blimp_only_npi_licensor_present": 1.0, "blimp_only_npi_scope": 1.0, "blimp_passive_1": 1.0, "blimp_passive_2": 1.0, "blimp_principle_A_c_command": 1.0, "blimp_principle_A_case_1": 1.0, "blimp_principle_A_case_2": 1.0, "blimp_principle_A_domain_1": 1.0, "blimp_principle_A_domain_2": 1.0, "blimp_principle_A_domain_3": 1.0, "blimp_principle_A_reconstruction": 1.0, "blimp_regular_plural_subject_verb_agreement_1": 1.0, "blimp_regular_plural_subject_verb_agreement_2": 1.0, "blimp_sentential_negation_npi_licensor_present": 1.0, "blimp_sentential_negation_npi_scope": 1.0, "blimp_sentential_subject_island": 1.0, "blimp_superlative_quantifiers_1": 1.0, "blimp_superlative_quantifiers_2": 1.0, "blimp_tough_vs_raising_1": 1.0, "blimp_tough_vs_raising_2": 1.0, "blimp_transitive": 1.0, "blimp_wh_island": 1.0, "blimp_wh_questions_object_gap": 1.0, "blimp_wh_questions_subject_gap": 1.0, "blimp_wh_questions_subject_gap_long_distance": 1.0, "blimp_wh_vs_that_no_gap": 1.0, "blimp_wh_vs_that_no_gap_long_distance": 1.0, "blimp_wh_vs_that_with_gap": 1.0, "blimp_wh_vs_that_with_gap_long_distance": 1.0, "commonsense_qa_text": "Yaml", "hellaswag": 1.0, "piqa": 1.0 } }, "rank_range": [ 18, 18 ], "release_date": "2026-06-01", "run_dir": "/home/max/PROJECTS/PythonProjects/CompactLMIndex/runs/gpt-s-1-4m/20260716T154647Z-v1.1", "source": "AxiomicLabs/GPT-S-1.4M", "track": "base", "updated_at": "2026-07-16T15:46:51.370873+00:00", "vocab_size": 4096 } ], "suite": "slm-benchmark-v1.1" }