diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4609bea09b55a4bc77a342165b92796454654470 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9eb51cfbcf588a249a71d3ad23287b30730c2391 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_sw_stem_tasks +task: global_mmlu_full_sw_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b0e9e67337c2fca6691c66a6731d34410eadbf4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb65e87ebc6d552982d19b70e2635771e2ec55bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b337d0ab8f133bea427f4085e0d413e01e6eabef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f44bfa0d9db24fb91020b0989eaa9ffd679ee4e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_sw_humanities_tasks +task: global_mmlu_full_sw_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eabd5a916f1858cd7c2c6c76fe5acd7c0cc34504 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_sw_humanities_tasks +task: global_mmlu_full_sw_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41c6445800ea3d53696ffbe3b3434b423f836618 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96edac9941a503dd29255f3d57047bc630b4a142 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_sw_humanities_tasks +task: global_mmlu_full_sw_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7cd19d35dc189953db225d5ace390fa65f9cba9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9434ae4c22f8213987379f93f014a1f36a41d173 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_sw_humanities_tasks +task: global_mmlu_full_sw_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf35b9c644fa3a50292cfd5c6a998a68b277e5af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7570e28899b60f7dd0c8a29542f28cdc802e986f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54c094db40c4f586ec3f11cba73f466954a47e46 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8d5a42c00d3bc9084dac75ab79a932c7159e521 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..79d51a58d088f82ccb5c7b58dce7e73654ddf3a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..523b1572bffef3758a6910a19c713687321255a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_sw_social_sciences_tasks +task: global_mmlu_full_sw_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43179ff8432de7e09c640a2f9e2c238f7fd74968 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_sw_other_tasks +task: global_mmlu_full_sw_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bef7b7f84ff14e8618af1580bf5c36f06198fcbb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/global_mmlu_full_sw_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _sw_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_sw_humanities_tasks +task: global_mmlu_full_sw_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/sw/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/_global_mmlu_full_te.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/_global_mmlu_full_te.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0bc967ded733ecf313badd12461bb249fbe6cbc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/_global_mmlu_full_te.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_te +task: + - global_mmlu_full_te_stem + - global_mmlu_full_te_other + - global_mmlu_full_te_social_sciences + - global_mmlu_full_te_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_abstract_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_abstract_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e922fd08d1826ceac13c9a4eb536a0f5f26443e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_abstract_algebra.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_abstract_algebra +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_abstract_algebra diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90e5618441a6f0284c7d1b86186c076528c717f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f036e60d50f8e4c2d98b032ac3082cd5e7ec9ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ccdb849a875c73c2030beb2cbd0d70f0586c76ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5022ce2981b9612ce5583b4d23fbd1a57a0a6e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd5219f073e7105a9177f562378d5bbca12070c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88dad05a2d9eacbef66427bc6e206a385da99204 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e8f37fc1dee6cdf5a64d3f52408342d8b15d4fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f05276252961898828e55ceaef16b767b1052a9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf008a6748dfe708ba6bcf35e700a81579751db5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97169e931a07143e4d1bf31abc353f5d24db506c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3edc89662958185b6737382b6c76fc0f814c7b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4c182d1d071d10898010a47da30b10f75ad40eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53b52f4d3ff79686c1018b6d8706685710316d02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f02170ff8d048cb3502e01966e6e77c3ae41f25 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c77d30aa2dda080c2314cda0ffe5e081276e7faf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f388a0606812de534f7496e7803954cecdd5036 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75d54d72bac17a0522dabbf2a7aff9830e275e16 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..383596ff024a193f5a94346128eb9cb0e6144a53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8db56a85168086d0de6160f13880b6f88eb96e2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd471b8d2e4cf4b3d992a03680c51f20bd817f0d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58f577ed1206932c53832a5b8a3f4bc8f139a126 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..400a3805f815dcb48835dfb5b2c8fce407c26350 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..694ddc304f0632d3b5b0403ae370694503ef440b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b900af19ccfe1aa095845f5a5c52db1678c9f9fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3492e7249129132ce1f659e3631c43eec49fc7e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48a2d75a6714e41b9f0466edb89caa4c409e3d94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e95f7ea19dd033356bac2d6385e3903c3fc71db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc44c1b0a1f1c616248f3b41da15ff2a645fddfb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7631419b702a0492a4c93b912b213ff3e879cc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c2c7862fbb8689e3ae5cdf3675b25b70ca395d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..718cd9fad0c2a1b5d966e4a0b4650ada3c3ddf71 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7bb9170ce69de22eece49f97c7cf65d2a6394d51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..123555383b3e423607fb92a85dddb5fe66a28547 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_te_stem_tasks +task: global_mmlu_full_te_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f092416ffbd3bf7b46a8e31ded9971e2c88de79e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15b84b4678fc5868c5e41f18ca00af67fda8dfed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f0730be97277f9e1d94caa47a5915dc8236a8c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_medical_genetics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_medical_genetics +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_medical_genetics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53487f558acb7e033c2ef7a2e7a71d362256a10b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fca8df9b7e1b62b0a6133511912754e4f72927cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d87f6b029e595bb6a81d110b69e6432b985176f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9348a76e2949054375aee37c3804225072ecf503 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8efe8d97b27ea77899c165997d641229742ef5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b702542ea997902dc6fbb8b83257e3fd034d3da7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..045b6e1ca5ac2fe20c04eaa51d335cee15f27a1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e5fa30806275cc5c615d58537fe7470e16adf45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4ede33f00c22cf9cf0363f15d2b6f742e5ef5eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cb1906d415899f46a26ecb76e36faae1d9e82615 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ac09ce0d6fcfe9ec0d38bcf616aa97532410da3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbb7bc7c81a4be3d388675c964b9f21319ed1427 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e080e0823b88d4d6ce06e007dddc376f2d50a621 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_te_social_sciences_tasks +task: global_mmlu_full_te_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f5e38a9eb80b76a0e48c6528aae8844dac25a04 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_te_other_tasks +task: global_mmlu_full_te_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4da26e3e17e0370a9f15543ebcbc31bca0384e7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/global_mmlu_full_te_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _te_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_te_humanities_tasks +task: global_mmlu_full_te_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/te/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa4ae63fa9d7a1243a4c6bcf11f4384f99912087 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr.yaml @@ -0,0 +1,11 @@ +group: global_mmlu_full_tr +task: + - global_mmlu_full_tr_stem + - global_mmlu_full_tr_other + - global_mmlu_full_tr_social_sciences + - global_mmlu_full_tr_humanities +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4dade158eb2ba0b672ed66d09b205db8cb97a67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_humanities.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_tr_humanities +task: + - global_mmlu_full_tr_humanities_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e80a5b9d8e30e0482db4d433181e2d3153c3b89f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_tr_other +task: + - global_mmlu_full_tr_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_social_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_social_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56fc20e14c019a09a3da4196c37fc0d4c6ab69d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_social_sciences.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_tr_social_sciences +task: + - global_mmlu_full_tr_social_sciences_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51f9bb3d90e89ec77e86d89d22e9468687e0b08f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_global_mmlu_full_tr_stem.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_tr_stem +task: + - global_mmlu_full_tr_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_tr_template_yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_tr_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e322bee67158028cfafaf5f066b2b06c5bf189d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/_tr_template_yaml @@ -0,0 +1,16 @@ +dataset_path: CohereForAI/Global-MMLU +dataset_name: tr +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\nD. {{option_d}}\nAnswer:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_anatomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_anatomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..444402254c31c5d20abca3a14610143a493a543f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_anatomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_anatomy +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_anatomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_astronomy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_astronomy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e85390bf1b17894295e767c8f780190cb3cdc286 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_astronomy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_astronomy +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_astronomy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_business_ethics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_business_ethics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b1afc9cf248c4896286603c7eaf40e7b5edf23c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_business_ethics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_business_ethics +tag: global_mmlu_full_tr_other_tasks +task: global_mmlu_full_tr_business_ethics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_clinical_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_clinical_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdfa69e6dc4ac181408c1803a777c619f551de00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_clinical_knowledge.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_clinical_knowledge +tag: global_mmlu_full_tr_other_tasks +task: global_mmlu_full_tr_clinical_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df43a67ce3c40d244893a9f723665da467bfc762 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af2b8b3e808a26211d054ffc36bb66bd61d54011 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_chemistry +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_college_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..622854f4d1723fe8e9acea54cb61a2ae505f055a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_computer_science +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_college_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..902bd9c1b4f8629f687057c0a3001912c3d375d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_mathematics +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_college_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b44d0d131520d179f9897eb08eeadce7483d6bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_tr_other_tasks +task: global_mmlu_full_tr_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27540d97f0f5e8629f49353977a2b37fea566096 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_computer_security.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_computer_security.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbcabeed0e45164a58dea395c6c662da5f6b2b5a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_computer_security.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_computer_security +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_computer_security diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_conceptual_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_conceptual_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..628a4fcf089c8b34623b684ea201d0e28bd9571f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_conceptual_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_conceptual_physics +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_conceptual_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6feb236f98053fe44b87c1d45976d58fda64c777 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_tr_social_sciences_tasks +task: global_mmlu_full_tr_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a2a8665cf779c170519fb6a1c7aabb38f6444ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ffc6dee79e3f7acde296c2a68ee3420b915f9a7b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77c189a0b016cdf5c7bb63469f71dd108fb6e622 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_tr_humanities_tasks +task: global_mmlu_full_tr_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a756d102e745702cc33e49169713a1bb242ed452 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_tr_other_tasks +task: global_mmlu_full_tr_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..51e7dd9e92f19f1c849b839f30fa78522254ed14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_tr_stem_tasks +task: global_mmlu_full_tr_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f88e98310ef3623c581a6586d4d2c4d6ab83a765 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_tr_social_sciences_tasks +task: global_mmlu_full_tr_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a28e110c3292adc977f8e13941ec391b718f81e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/tr/global_mmlu_full_tr_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _tr_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_tr_social_sciences_tasks +task: global_mmlu_full_tr_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad12bde64e609adcf43a18057dd13dd95d79ffd6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_yo_other_tasks +task: global_mmlu_full_yo_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..198f227b893a3758d463ac50d867e6ee3355b76c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_yo_other_tasks +task: global_mmlu_full_yo_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1956c87cdf7676dad84c956e5d32ace13c39841 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_yo_social_sciences_tasks +task: global_mmlu_full_yo_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c6c2b8c1c5444066af0541751816b9198caa307 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_yo_social_sciences_tasks +task: global_mmlu_full_yo_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a12c4abd3e44cc1e7b256efc95bdcf59ec2a6308 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_yo_social_sciences_tasks +task: global_mmlu_full_yo_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e5747900bca8d2a1c00ae902353f5d1938e03617 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_yo_social_sciences_tasks +task: global_mmlu_full_yo_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..493dda39641e29de5d42257e4837dd196aeed92d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_yo_social_sciences_tasks +task: global_mmlu_full_yo_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..420b1b01c8bbe069eb3b397b29e8a3ffef620a21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_yo_other_tasks +task: global_mmlu_full_yo_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0964b30717ccf68ee13a8be218f52cdbf56d078 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/global_mmlu_full_yo_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _yo_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_yo_humanities_tasks +task: global_mmlu_full_yo_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/yo/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/_global_mmlu_full_zh_other.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/_global_mmlu_full_zh_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..98d4ed5e613fee710e2f3fe210c4da2abddd7e9e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/_global_mmlu_full_zh_other.yaml @@ -0,0 +1,8 @@ +group: global_mmlu_full_zh_other +task: + - global_mmlu_full_zh_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba2031feff45c3c7db26a8e30207cdff0bc81a1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_college_biology +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_college_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..523d6b3075fda6eb41abb160dad2c9a746fa9599 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_college_medicine +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_college_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a08214f7fe96bedab8f3b1bb55fbb161b711119 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_college_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_college_physics +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_college_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_econometrics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_econometrics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf920112179d6d81df9c20512beb31a3422eaba4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_econometrics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_econometrics +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_econometrics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b30acad7621b2c8339eaa16c3bdb55fb6b9929a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_electrical_engineering.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_electrical_engineering +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_electrical_engineering diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_elementary_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_elementary_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3b108c421370e7f86d5aa9bb44a087f029e1aa6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_elementary_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_elementary_mathematics +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_elementary_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_formal_logic.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_formal_logic.yaml new file mode 100644 index 0000000000000000000000000000000000000000..647755991d84de30207ae9010f1a89f2ebc7d4a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_formal_logic.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_formal_logic +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_formal_logic diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_global_facts.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_global_facts.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07d390aab2f0c30f68df1b6df744404902cb5d14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_global_facts.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_global_facts +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_global_facts diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..28b2bdaa54cc1ab9dcad563983a4ab84abb25460 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_biology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_biology +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_biology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d084034da2abd8306485999ff67596a618c742f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_chemistry.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_chemistry +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_chemistry diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6232ef607ecfcd8922518d269909d0a1a1b9dff4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_computer_science.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_computer_science +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_computer_science diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_european_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_european_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70e3e52bc9c834410672b8d2b534e35c4f4470b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_european_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_european_history +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_high_school_european_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_geography.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_geography.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe6cb91391b8721579e1a9a1d0cfd4340d6fcd1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_geography.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_geography +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_high_school_geography diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_government_and_politics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_government_and_politics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfa7213a4b5e4f614c3cee431a3626875eb102a5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_government_and_politics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_government_and_politics +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_high_school_government_and_politics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_macroeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_macroeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca0b7ad86f6facaca944eb360f7f5f6ed3d5c0f9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_macroeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_macroeconomics +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_high_school_macroeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_mathematics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_mathematics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38868e969bf5d308b624dd6f4015605405aa9951 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_mathematics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_mathematics +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_mathematics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_microeconomics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_microeconomics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b79237d2f04a713f51bf1ec59bf5b6d7f0b5179d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_microeconomics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_microeconomics +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_high_school_microeconomics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_physics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_physics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6355da2f8e7196062ed7eda14ae58449daab18e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_physics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_physics +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_physics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f223886737a2157807ca44f74c717d6955d3f66a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_psychology +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_high_school_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_statistics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_statistics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9aac20977c68333b736f60f548ef4ca506f04960 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_statistics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_statistics +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_high_school_statistics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_us_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_us_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47d8355f5bfe5d7905efd9e0b8f3c80babc88700 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_us_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_us_history +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_high_school_us_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_world_history.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_world_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1f6671f2d183423812f1053d1e1052fc4a9aea0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_high_school_world_history.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_high_school_world_history +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_high_school_world_history diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_aging.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_aging.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d6941ff7b2f110c7ea63fc623622f967802c3c8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_aging.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_human_aging +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_human_aging diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_sexuality.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_sexuality.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee228b22d38c2c25a361f745ebdbafcbdfe58743 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_human_sexuality.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_human_sexuality +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_human_sexuality diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_international_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_international_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07b1ebd16f4b5cfca32969742b2691c6ee3839f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_international_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_international_law +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_international_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_jurisprudence.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_jurisprudence.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab10ffac5daa7950533669e07708f827099ee429 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_jurisprudence.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_jurisprudence +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_jurisprudence diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_logical_fallacies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_logical_fallacies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..451260b55f483511bd543a401881c29c3a9e6815 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_logical_fallacies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_logical_fallacies +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_logical_fallacies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_machine_learning.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_machine_learning.yaml new file mode 100644 index 0000000000000000000000000000000000000000..508d14f64d8f5fe198f36550d697f91100297c14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_machine_learning.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_machine_learning +tag: global_mmlu_full_zh_stem_tasks +task: global_mmlu_full_zh_machine_learning diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_management.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9db0b32bb1f162c4d84b2a967465d1de9dd3fb17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_management.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_management +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_management diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7142ce49f5cb9f0deeae7e4b1d10a291c364220 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_marketing.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_marketing +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_marketing diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_medical_genetics.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_medical_genetics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..220530907d9c9f952163216967fe7545dde8a072 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_medical_genetics.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_medical_genetics +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_medical_genetics diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_miscellaneous.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_miscellaneous.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b479c9b0cfd4160c92a4a71aae143789fc9c283 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_miscellaneous.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_miscellaneous +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_miscellaneous diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_disputes.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_disputes.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58d13a9944ad81bef3c761b55cede58c60571368 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_disputes.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_moral_disputes +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_moral_disputes diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_scenarios.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_scenarios.yaml new file mode 100644 index 0000000000000000000000000000000000000000..95d91dfdebfd1f84481228d8fccee406399a43f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_moral_scenarios.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_moral_scenarios +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_moral_scenarios diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_nutrition.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_nutrition.yaml new file mode 100644 index 0000000000000000000000000000000000000000..57452a39b09498bc0f1c753dba424feb56bbef21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_nutrition.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_nutrition +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_nutrition diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_philosophy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_philosophy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20e237b24cdd4992e4c5e0c6ee1b191c5bc5a2b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_philosophy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_philosophy +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_philosophy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_prehistory.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_prehistory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..56358fe789834a394bea785e64dc49183cf080cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_prehistory.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_prehistory +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_prehistory diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..630681abf67acca1ce5b1bcc869f16229f5699ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_accounting.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_professional_accounting +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_professional_accounting diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_law.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e48f35cb38bad9514dad8bf685dc74dfc49951a2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_law.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_professional_law +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_professional_law diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_medicine.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_medicine.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f75432cd7d9c9d8c46050b669ac80f040e821df3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_medicine.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_professional_medicine +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_professional_medicine diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fbbf45ad76940af904ca88b0ed6705ae5cf015b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_professional_psychology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_professional_psychology +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_professional_psychology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_public_relations.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_public_relations.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f760d2a29df46b1cfd1737f3304925477ed09a49 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_public_relations.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_public_relations +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_public_relations diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_security_studies.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_security_studies.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1dafaf5d754525a6ef06d5d3a6a61196a7c4982b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_security_studies.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_security_studies +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_security_studies diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..549f4ef156f784a95acd6a6d6bec93398944fb5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_sociology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_sociology +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_sociology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_us_foreign_policy.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_us_foreign_policy.yaml new file mode 100644 index 0000000000000000000000000000000000000000..597dcfa1cca601c17fce1facc62dcb3830220869 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_us_foreign_policy.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_us_foreign_policy +tag: global_mmlu_full_zh_social_sciences_tasks +task: global_mmlu_full_zh_us_foreign_policy diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_virology.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_virology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1984c6b00249308ba504b84a754025b38c422cac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_virology.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_virology +tag: global_mmlu_full_zh_other_tasks +task: global_mmlu_full_zh_virology diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_world_religions.yaml b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_world_religions.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa15c0cb282b9ff2bab168748b13a50b0de85d78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/global_mmlu_full_zh_world_religions.yaml @@ -0,0 +1,5 @@ +# Generated by _generate_configs.py +include: _zh_template_yaml +process_docs: !function utils.process_world_religions +tag: global_mmlu_full_zh_humanities_tasks +task: global_mmlu_full_zh_world_religions diff --git a/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/utils.py b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..7df72cb061f0fecba46e15ff9b57552817979afb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/global_mmlu/full/zh/utils.py @@ -0,0 +1,73 @@ +from functools import partial + + +SUBJECTS = [ + "abstract_algebra", + "anatomy", + "astronomy", + "business_ethics", + "clinical_knowledge", + "college_biology", + "college_chemistry", + "college_computer_science", + "college_mathematics", + "college_medicine", + "college_physics", + "computer_security", + "conceptual_physics", + "econometrics", + "electrical_engineering", + "elementary_mathematics", + "formal_logic", + "global_facts", + "high_school_biology", + "high_school_chemistry", + "high_school_computer_science", + "high_school_european_history", + "high_school_geography", + "high_school_government_and_politics", + "high_school_macroeconomics", + "high_school_mathematics", + "high_school_microeconomics", + "high_school_physics", + "high_school_psychology", + "high_school_statistics", + "high_school_us_history", + "high_school_world_history", + "human_aging", + "human_sexuality", + "international_law", + "jurisprudence", + "logical_fallacies", + "machine_learning", + "management", + "marketing", + "medical_genetics", + "miscellaneous", + "moral_disputes", + "moral_scenarios", + "nutrition", + "philosophy", + "prehistory", + "professional_accounting", + "professional_law", + "professional_medicine", + "professional_psychology", + "public_relations", + "security_studies", + "sociology", + "us_foreign_policy", + "virology", + "world_religions", +] + + +def process_docs(dataset, subject): + return dataset.filter(lambda x: x["subject"] == subject) + + +process_functions = { + f"process_{subject}": partial(process_docs, subject=subject) for subject in SUBJECTS +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/README.md b/lm-evaluation-harness/lm_eval/tasks/glue/README.md new file mode 100644 index 0000000000000000000000000000000000000000..91c35cb4a4599ef74ac2d36586c2eaca43916263 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/README.md @@ -0,0 +1,76 @@ +# GLUE +**NOTE**: GLUE benchmark tasks do not provide publicly accessible labels for their test sets, so we default to the validation sets for all sub-tasks. + +### Paper + +Title: `GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding` + +Abstract: https://openreview.net/pdf?id=rJ4km2R5t7 + +The General Language Understanding Evaluation (GLUE) benchmark is a collection of +resources for training, evaluating, and analyzing natural language understanding +systems. GLUE consists of: +- A benchmark of nine sentence- or sentence-pair language understanding tasks built +on established existing datasets and selected to cover a diverse range of dataset +sizes, text genres, and degrees of difficulty, and +- A diagnostic dataset designed to evaluate and analyze model performance with +respect to a wide range of linguistic phenomena found in natural language. + +Homepage: https://gluebenchmark.com/ + +### Citation + +``` +@inproceedings{wang-etal-2018-glue, + title = "{GLUE}: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding", + author = "Wang, Alex and + Singh, Amanpreet and + Michael, Julian and + Hill, Felix and + Levy, Omer and + Bowman, Samuel", + booktitle = "Proceedings of the 2018 {EMNLP} Workshop {B}lackbox{NLP}: Analyzing and Interpreting Neural Networks for {NLP}", + month = nov, + year = "2018", + address = "Brussels, Belgium", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/W18-5446", + doi = "10.18653/v1/W18-5446", + pages = "353--355", + abstract = "Human ability to understand language is \textit{general, flexible, and robust}. In contrast, most NLU models above the word level are designed for a specific task and struggle with out-of-domain data. If we aspire to develop models with understanding beyond the detection of superficial correspondences between inputs and outputs, then it is critical to develop a unified model that can execute a range of linguistic tasks across different domains. To facilitate research in this direction, we present the General Language Understanding Evaluation (GLUE, gluebenchmark.com): a benchmark of nine diverse NLU tasks, an auxiliary dataset for probing models for understanding of specific linguistic phenomena, and an online platform for evaluating and comparing models. For some benchmark tasks, training data is plentiful, but for others it is limited or does not match the genre of the test set. GLUE thus favors models that can represent linguistic knowledge in a way that facilitates sample-efficient learning and effective knowledge-transfer across tasks. While none of the datasets in GLUE were created from scratch for the benchmark, four of them feature privately-held test data, which is used to ensure that the benchmark is used fairly. We evaluate baselines that use ELMo (Peters et al., 2018), a powerful transfer learning technique, as well as state-of-the-art sentence representation models. The best models still achieve fairly low absolute scores. Analysis with our diagnostic dataset yields similarly weak performance over all phenomena tested, with some exceptions.", +} +``` + +### Groups, Tags, and Tasks + +#### Groups + +None. + +#### Tags + +* `glue`: Run all Glue subtasks. + +#### Tasks + +* `cola` +* `mnli` +* `mrpc` +* `qnli` +* `qqp` +* `rte` +* `sst` +* `wnli` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/cola/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/cola/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f79e5e8f6403014e790726d8d66eac86629ec90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/cola/default.yaml @@ -0,0 +1,16 @@ +tag: glue +task: cola +dataset_path: glue +dataset_name: cola +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "{{sentence}}\nQuestion: Does this sentence make sense?\nAnswer:" +doc_to_target: label +doc_to_choice: ["no", "yes"] +should_decontaminate: true +doc_to_decontamination_query: sentence +metric_list: + - metric: mcc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/mnli/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..654f61231fa51f7688daf8063514656aa7e29283 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/default.yaml @@ -0,0 +1,14 @@ +tag: glue +task: mnli +dataset_path: glue +dataset_name: mnli +output_type: multiple_choice +training_split: train +validation_split: validation_matched +doc_to_text: !function utils.doc_to_text +doc_to_target: label +doc_to_choice: ["True", "Neither", "False"] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/mnli/mismatch.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/mismatch.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e9b49bcd423ce43bf87f044c75a01e75f44d3d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/mismatch.yaml @@ -0,0 +1,3 @@ +include: default.yaml +task: mnli_mismatch +validation_split: validation_mismatched diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/mnli/utils.py b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..2d5fdaec2905ac7cf95ac3e50f1d12c728f59c37 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/mnli/utils.py @@ -0,0 +1,6 @@ +def doc_to_text(doc) -> str: + return "{}\nQuestion: {} True, False or Neither?\nAnswer:".format( + doc["premise"], + doc["hypothesis"].strip() + + ("" if doc["hypothesis"].strip().endswith(".") else "."), + ) diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/mrpc/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/mrpc/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cdbb8bbd0f4f7c756dfc2ee436f76513400c154 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/mrpc/default.yaml @@ -0,0 +1,15 @@ +tag: glue +task: mrpc +dataset_path: glue +dataset_name: mrpc +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "Sentence 1: {{sentence1}}\nSentence 2: {{sentence2}}\nQuestion: Do both sentences mean the same thing?\nAnswer:" +doc_to_target: label +doc_to_choice: ["no", "yes"] +metric_list: + - metric: acc + - metric: f1 +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/qnli/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/qnli/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e069209e27774cbcb4bba67862c6b58ad8113b2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/qnli/default.yaml @@ -0,0 +1,14 @@ +tag: glue +task: qnli +dataset_path: glue +dataset_name: qnli +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "{{question}}\n{{sentence}}\nQuestion: Does this response answer the question?\nAnswer:" +doc_to_target: label +doc_to_choice: ["yes", "no"] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/qqp/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/qqp/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f76da063e635c8aaa31e9f2d365655f2ad1df091 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/qqp/default.yaml @@ -0,0 +1,15 @@ +tag: glue +task: qqp +dataset_path: glue +dataset_name: qqp +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "Question 1: {{question1}}\nQuestion 2: {{question2}}\nQuestion: Do both questions ask the same thing?\nAnswer:" +doc_to_target: label +doc_to_choice: ["no", "yes"] +metric_list: + - metric: acc + - metric: f1 +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/rte/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/rte/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..216c5210622e997b5623418271993934dbfa6e09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/rte/default.yaml @@ -0,0 +1,14 @@ +tag: glue +task: rte +dataset_path: glue +dataset_name: rte +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "{{sentence1}}\nQuestion: {{sentence2}} True or False?\nAnswer:" +doc_to_target: label +doc_to_choice: ["True", "False"] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/sst2/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/sst2/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..160e3e08a205bc8c369852b54d926c59a34d0f6b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/sst2/default.yaml @@ -0,0 +1,14 @@ +tag: glue +task: sst2 +dataset_path: glue +dataset_name: sst2 +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "{{sentence}}\nQuestion: Is this sentence positive or negative?\nAnswer:" +doc_to_target: label +doc_to_choice: ["negative", "positive"] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/glue/wnli/default.yaml b/lm-evaluation-harness/lm_eval/tasks/glue/wnli/default.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63966e4c8be81b78eca1b4b1540583199fd42ee2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/glue/wnli/default.yaml @@ -0,0 +1,14 @@ +tag: glue +task: wnli +dataset_path: glue +dataset_name: wnli +output_type: multiple_choice +training_split: train +validation_split: validation +doc_to_text: "{{sentence1}}\nQuestion: {{sentence2}} True or False?\nAnswer:" +doc_to_target: label +doc_to_choice: ["False", "True"] +metric_list: + - metric: acc +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/README.md b/lm-evaluation-harness/lm_eval/tasks/gpqa/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7c148d9fbec8be41fd89a01aa8590deabd2c4cad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/README.md @@ -0,0 +1,59 @@ +# GPQA + +### Paper + +Title: GPQA: A Graduate-Level Google-Proof Q&A Benchmark + +Abstract: https://arxiv.org/abs/2311.12022 + +We present GPQA, a challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. We ensure that the questions are high-quality and extremely difficult: experts who have or are pursuing PhDs in the corresponding domains reach 65% accuracy (74% when discounting clear mistakes the experts identified in retrospect), while highly skilled non-expert validators only reach 34% accuracy, despite spending on average over 30 minutes with unrestricted access to the web (i.e., the questions are “Google-proof”). The questions are also difficult for state-of-the-art AI systems, with our strongest GPT-4–based baseline achieving 39% accuracy. If we are to use future AI systems to help us answer very hard questions—for example, when developing new scientific knowledge—we need to develop *scalable oversight* methods that enable humans to supervise their outputs, which may be difficult even if the supervisors are themselves skilled and knowledgeable. The difficulty of GPQA both for skilled non-experts and frontier AI systems should enable realistic scalable oversight experiments, which we hope can help devise ways for human experts to reliably get truthful information from AI systems that surpass human capabilities. + +Homepage: `https://github.com/idavidrein/gpqa/tree/main` + +### Citation + +``` +@misc{rein2023gpqa, + title={GPQA: A Graduate-Level Google-Proof Q&A Benchmark}, + author={David Rein and Betty Li Hou and Asa Cooper Stickland and Jackson Petty and Richard Yuanzhe Pang and Julien Dirani and Julian Michael and Samuel R. Bowman}, + year={2023}, + eprint={2311.12022}, + archivePrefix={arXiv}, + primaryClass={cs.AI} +} +``` + +This dataset is gated, so you will have to accept the terms of use at https://huggingface.co/datasets/Idavidrein/gpqa and login via `huggingface-cli login` using your HF Hub token before running this task. + +### Groups, Tags, and Tasks + +#### Groups + +None + +#### Tags + +* `gpqa`: runs all GPQA variants. + +#### Tasks + +* `gpqa_{main, diamond, extended}_zeroshot` +* `gpqa_{main, diamond, extended}_n_shot` +* `gpqa_{main, diamond, extended}_generative_n_shot` +* `gpqa_{main, diamond, extended}_cot_zeroshot` +* `gpqa_{main, diamond, extended}_cot_n_shot` + +### Checklist + +For adding novel benchmarks/datasets to the library: + +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: + +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..73ccb876a449a1e8eda5984d977194f6b0c064d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_generate_configs.py @@ -0,0 +1,26 @@ +import yaml +from tqdm import tqdm + + +def main() -> None: + subset = ["extended", "diamond", "main"] + setting = "cot_n_shot" + for task in tqdm(subset): + file_name = f"gpqa_{task}_{setting}.yaml" + try: + with open(f"{file_name}", "w") as f: + f.write("# Generated by _generate_configs.py\n") + yaml.dump( + { + "include": f"_gpqa_{setting}_yaml", + "task": f"gpqa_{task}_{setting}", + "dataset_name": f"gpqa_{task}", + }, + f, + ) + except FileExistsError: + pass + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_gpqa_cot_n_shot_yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_gpqa_cot_n_shot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..97c0603bcc94f0c689269ea9859b62bdfab7644e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/_gpqa_cot_n_shot_yaml @@ -0,0 +1,38 @@ +dataset_path: Idavidrein/gpqa +tag: gpqa +output_type: generate_until +process_docs: !function utils.process_docs +training_split: train +# Because huggingface dataset only has train split +validation_split: train +test_split: null +description: "Here are some example questions from experts. Answer the final question yourself, following the format of the previous questions exactly.\n" +doc_to_text: "Question: {{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nLet's think step by step: " +doc_to_target: answer +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "(?<=The answer is )(.*)(?=.)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "multi_choice_regex" + group_select: -1 + ignore_case: true + ignore_punctuation: true + regex_pattern: "(\\([A-Z]\\))" + - function: "take_first" +generation_kwargs: + until: + - "" + do_sample: false + temperature: 0.0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_diamond_cot_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_diamond_cot_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24e5f4f90f1f770f9f792e4aeef51e08d3aa08d9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_diamond_cot_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_diamond +include: _gpqa_cot_n_shot_yaml +task: gpqa_diamond_cot_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_extended_cot_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_extended_cot_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..002ede9a82110e3679bf3e1e958ded4342e408e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_extended_cot_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_extended +include: _gpqa_cot_n_shot_yaml +task: gpqa_extended_cot_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_main_cot_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_main_cot_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..916b6ea06a2e22042344b668191adbb3c91c4e75 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/gpqa_main_cot_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_main +include: _gpqa_cot_n_shot_yaml +task: gpqa_main_cot_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/utils.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..96bcd52b140fd0a5896f55c0a52ea2fd5453fd53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_n_shot/utils.py @@ -0,0 +1,39 @@ +import random +import re + +import datasets + + +def preprocess(text): + if text is None: + return " " + text = text.strip() + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + choices = [ + preprocess(doc["Incorrect Answer 1"]), + preprocess(doc["Incorrect Answer 2"]), + preprocess(doc["Incorrect Answer 3"]), + preprocess(doc["Correct Answer"]), + ] + + random.shuffle(choices) + correct_answer_index = choices.index(preprocess(doc["Correct Answer"])) + + out_doc = { + "choice1": choices[0], + "choice2": choices[1], + "choice3": choices[2], + "choice4": choices[3], + "choices": [choices[0], choices[1], choices[2], choices[3]], + "answer": f"({chr(65 + correct_answer_index)})", + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..bda00784cc2fa26b5f0d488cf7b6aea37243353d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_generate_configs.py @@ -0,0 +1,26 @@ +import yaml +from tqdm import tqdm + + +def main() -> None: + subset = ["extended", "diamond", "main"] + setting = "cot_zeroshot" + for task in tqdm(subset): + file_name = f"gpqa_{task}_{setting}.yaml" + try: + with open(f"{file_name}", "w") as f: + f.write("# Generated by _generate_configs.py\n") + yaml.dump( + { + "include": f"_gpqa_{setting}_yaml", + "task": f"gpqa_{task}_{setting}", + "dataset_name": f"gpqa_{task}", + }, + f, + ) + except FileExistsError: + pass + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_gpqa_cot_zeroshot_yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_gpqa_cot_zeroshot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c487a8c4a3e3806bfa265fa7dc7a3f897ddedff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/_gpqa_cot_zeroshot_yaml @@ -0,0 +1,38 @@ +dataset_path: Idavidrein/gpqa +tag: gpqa +output_type: generate_until +process_docs: !function utils.process_docs +training_split: train +# Because huggingface dataset only has train split +validation_split: train +test_split: null +doc_to_text: "What is the correct answer to this question:{{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nLet's think step by step: " +doc_to_target: answer +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "(?<=The answer is )(.*)(?=.)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "multi_choice_regex" + group_select: -1 + ignore_case: true + ignore_punctuation: true + regex_pattern: "(\\([A-Z]\\))" + - function: "take_first" +generation_kwargs: + until: + - "" + do_sample: false + temperature: 0.0 +num_fewshot: 0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e6a840fa1815096f5fa180ed06223e3523a06214 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_diamond_cot_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_diamond +include: _gpqa_cot_zeroshot_yaml +task: gpqa_diamond_cot_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_extended_cot_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_extended_cot_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f542a6148f231e2d7e7e2a5a3437047459e3856 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_extended_cot_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_extended +include: _gpqa_cot_zeroshot_yaml +task: gpqa_extended_cot_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_main_cot_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_main_cot_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c14604854294c4551e2602e573488c6a7fef254 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/gpqa_main_cot_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_main +include: _gpqa_cot_zeroshot_yaml +task: gpqa_main_cot_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/utils.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..96bcd52b140fd0a5896f55c0a52ea2fd5453fd53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/cot_zeroshot/utils.py @@ -0,0 +1,39 @@ +import random +import re + +import datasets + + +def preprocess(text): + if text is None: + return " " + text = text.strip() + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + choices = [ + preprocess(doc["Incorrect Answer 1"]), + preprocess(doc["Incorrect Answer 2"]), + preprocess(doc["Incorrect Answer 3"]), + preprocess(doc["Correct Answer"]), + ] + + random.shuffle(choices) + correct_answer_index = choices.index(preprocess(doc["Correct Answer"])) + + out_doc = { + "choice1": choices[0], + "choice2": choices[1], + "choice3": choices[2], + "choice4": choices[3], + "choices": [choices[0], choices[1], choices[2], choices[3]], + "answer": f"({chr(65 + correct_answer_index)})", + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..e2c011ea02d25ca1d3550210f4a4644c97fa52c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_generate_configs.py @@ -0,0 +1,26 @@ +import yaml +from tqdm import tqdm + + +def main() -> None: + subset = ["extended", "diamond", "main"] + setting = "generative_n_shot" + for task in tqdm(subset): + file_name = f"gpqa_{task}_{setting}.yaml" + try: + with open(f"{file_name}", "w") as f: + f.write("# Generated by _generate_configs.py\n") + yaml.dump( + { + "include": f"_gpqa_{setting}_yaml", + "task": f"gpqa_{task}_{setting}", + "dataset_name": f"gpqa_{task}", + }, + f, + ) + except FileExistsError: + pass + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_gpqa_generative_n_shot_yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_gpqa_generative_n_shot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..f43a9a414cb4e53e7d5e83787ae6c1e5de109111 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/_gpqa_generative_n_shot_yaml @@ -0,0 +1,39 @@ +dataset_path: Idavidrein/gpqa +tag: gpqa +output_type: generate_until +process_docs: !function utils.process_docs +training_split: train +# Because huggingface dataset only has train split +validation_split: train +test_split: null +description: "Here are some example questions from experts. Answer the final question yourself, following the format of the previous questions exactly.\n" +doc_to_text: "Question: {{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nAnswer:" +doc_to_target: answer +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "(?<=The answer is )(.*)(?=.)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "multi_choice_regex" + group_select: -1 + ignore_case: true + ignore_punctuation: true + regex_pattern: "(\\([A-Z]\\))" + - function: "take_first" +generation_kwargs: + until: + - "" + - "Question:" + - "<|im_end|>" + temperature: 0.0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_diamond_generative_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_diamond_generative_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a42094e8ba8ef6037820255b74a8830d550b8a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_diamond_generative_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_diamond +include: _gpqa_generative_n_shot_yaml +task: gpqa_diamond_generative_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_extended_generative_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_extended_generative_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc40c2d97684c50b3992f5adf894ebe0c138b4ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_extended_generative_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_extended +include: _gpqa_generative_n_shot_yaml +task: gpqa_extended_generative_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_main_generative_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_main_generative_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..865f3cb5efa3d4b8641843cfde7db3c95bd8b8b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/gpqa_main_generative_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_main +include: _gpqa_generative_n_shot_yaml +task: gpqa_main_generative_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/utils.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..96bcd52b140fd0a5896f55c0a52ea2fd5453fd53 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/generative/utils.py @@ -0,0 +1,39 @@ +import random +import re + +import datasets + + +def preprocess(text): + if text is None: + return " " + text = text.strip() + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + choices = [ + preprocess(doc["Incorrect Answer 1"]), + preprocess(doc["Incorrect Answer 2"]), + preprocess(doc["Incorrect Answer 3"]), + preprocess(doc["Correct Answer"]), + ] + + random.shuffle(choices) + correct_answer_index = choices.index(preprocess(doc["Correct Answer"])) + + out_doc = { + "choice1": choices[0], + "choice2": choices[1], + "choice3": choices[2], + "choice4": choices[3], + "choices": [choices[0], choices[1], choices[2], choices[3]], + "answer": f"({chr(65 + correct_answer_index)})", + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..c01f208e767cb813e6d2116caf74c3d0b2fccfb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_generate_configs.py @@ -0,0 +1,26 @@ +import yaml +from tqdm import tqdm + + +def main() -> None: + subset = ["extended", "diamond", "main"] + + for task in tqdm(subset): + file_name = f"gpqa_{task}_n_shot.yaml" + try: + with open(f"{file_name}", "w") as f: + f.write("# Generated by _generate_configs.py\n") + yaml.dump( + { + "include": "_gpqa_n_shot_yaml", + "task": f"gpqa_{task}_n_shot", + "dataset_name": f"gpqa_{task}", + }, + f, + ) + except FileExistsError: + pass + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_gpqa_n_shot_yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_gpqa_n_shot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..8406f8aabfa9d10eec18ef7a8565b6393a0bfc03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/_gpqa_n_shot_yaml @@ -0,0 +1,21 @@ +dataset_path: Idavidrein/gpqa +tag: gpqa +output_type: multiple_choice +process_docs: !function utils.process_docs +training_split: train +# Because huggingface dataset only has train split +validation_split: train +test_split: null +description: "Here are some example questions from experts. Answer the final question yourself, following the format of the previous questions exactly.\n" +doc_to_text: "Question: {{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nAnswer:" +doc_to_target: answer +doc_to_choice: ["(A)", "(B)", "(C)", "(D)"] +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_diamond_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_diamond_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3043a7e53647ff72d535abc113dfccebaa1bd43c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_diamond_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_diamond +include: _gpqa_n_shot_yaml +task: gpqa_diamond_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_extended_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_extended_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d16b505b355bccb3d6fd70eb16b307c12d06a09 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_extended_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_extended +include: _gpqa_n_shot_yaml +task: gpqa_extended_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_main_n_shot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_main_n_shot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e5f3e9532ab41c0158409e6afb47393806c4177 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/gpqa_main_n_shot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_main +include: _gpqa_n_shot_yaml +task: gpqa_main_n_shot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/utils.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..e0b886d2879216094214ce534438e4db0c5e60f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/n_shot/utils.py @@ -0,0 +1,41 @@ +import random +import re + +import datasets + + +def preprocess(text): + if text is None: + return " " + text = text.strip() + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +rng = random.Random(42) + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + choices = [ + preprocess(doc["Incorrect Answer 1"]), + preprocess(doc["Incorrect Answer 2"]), + preprocess(doc["Incorrect Answer 3"]), + preprocess(doc["Correct Answer"]), + ] + + rng.shuffle(choices) + correct_answer_index = choices.index(preprocess(doc["Correct Answer"])) + + out_doc = { + "choice1": choices[0], + "choice2": choices[1], + "choice3": choices[2], + "choice4": choices[3], + "answer": f"({chr(65 + correct_answer_index)})", + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_generate_configs.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_generate_configs.py new file mode 100644 index 0000000000000000000000000000000000000000..79afbd6f1d8d4b2eb54455d734f6245357580bd3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_generate_configs.py @@ -0,0 +1,26 @@ +import yaml +from tqdm import tqdm + + +def main() -> None: + subset = ["extended", "diamond", "main"] + setting = "zeroshot" + for task in tqdm(subset): + file_name = f"gpqa_{task}_{setting}.yaml" + try: + with open(f"{file_name}", "w") as f: + f.write("# Generated by _generate_configs.py\n") + yaml.dump( + { + "include": f"_gpqa_{setting}_yaml", + "task": f"gpqa_{task}_{setting}", + "dataset_name": f"gpqa_{task}", + }, + f, + ) + except FileExistsError: + pass + + +if __name__ == "__main__": + main() diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_gpqa_zeroshot_yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_gpqa_zeroshot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..500f1921bec3db0d1282b8501b7a0841ebbb79c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/_gpqa_zeroshot_yaml @@ -0,0 +1,21 @@ +dataset_path: Idavidrein/gpqa +tag: gpqa +output_type: multiple_choice +process_docs: !function utils.process_docs +training_split: train +# Because huggingface dataset only has train split +validation_split: train +test_split: null +doc_to_text: "What is the correct answer to this question:{{Question}}\nChoices:\n(A) {{choice1}}\n(B) {{choice2}}\n(C) {{choice3}}\n(D) {{choice4}}\nAnswer:" +doc_to_target: answer +doc_to_choice: ["(A)", "(B)", "(C)", "(D)"] +num_fewshot: 0 +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_diamond_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_diamond_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3a7921c30b3ff09e82aacb4c0e915010f698966 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_diamond_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_diamond +include: _gpqa_zeroshot_yaml +task: gpqa_diamond_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_extended_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_extended_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e7347f11154351ad4560200a3f3bf54106a1a8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_extended_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_extended +include: _gpqa_zeroshot_yaml +task: gpqa_extended_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_main_zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_main_zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a8d7fb59025d148130f2a468cb1bbdfad959102 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/gpqa_main_zeroshot.yaml @@ -0,0 +1,4 @@ +# Generated by _generate_configs.py +dataset_name: gpqa_main +include: _gpqa_zeroshot_yaml +task: gpqa_main_zeroshot diff --git a/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/utils.py b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..c2317e02efd132aea27ec8c8fad284df55ccd382 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gpqa/zeroshot/utils.py @@ -0,0 +1,38 @@ +import random +import re + +import datasets + + +def preprocess(text): + if text is None: + return " " + text = text.strip() + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + choices = [ + preprocess(doc["Incorrect Answer 1"]), + preprocess(doc["Incorrect Answer 2"]), + preprocess(doc["Incorrect Answer 3"]), + preprocess(doc["Correct Answer"]), + ] + + random.shuffle(choices) + correct_answer_index = choices.index(preprocess(doc["Correct Answer"])) + + out_doc = { + "choice1": choices[0], + "choice2": choices[1], + "choice3": choices[2], + "choice4": choices[3], + "answer": f"({chr(65 + correct_answer_index)})", + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/groundcocoa/README.md b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/README.md new file mode 100644 index 0000000000000000000000000000000000000000..8944afffa2bc0318ed8235306bc7ef72215b7935 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/README.md @@ -0,0 +1,50 @@ +# GroundCocoa + +### Paper + +Title: `GroundCocoa: A Benchmark for Evaluating Compositional & Conditional Reasoning in Language Models` + +Abstract: https://arxiv.org/abs/2404.04237 + +The rapid progress of large language models (LLMs) has seen them excel and frequently surpass human performance on standard benchmarks. This has enabled many downstream applications, such as LLM agents, to rely on their reasoning to address complex task requirements. However, LLMs are known to unexpectedly falter in simple tasks and under seemingly straightforward circumstances - underscoring the need for better and more diverse evaluation setups to measure their true capabilities. To this end, we choose to study compositional and conditional reasoning, two aspects that are central to human cognition, and introduce GroundCocoa - a lexically diverse benchmark connecting these reasoning skills to the real-world problem of flight booking. Our task involves aligning detailed user preferences with available flight options presented in a multiple-choice format. Results indicate a significant disparity in performance among current state-of-the-art LLMs with even the best performing model, GPT-4 Turbo, not exceeding 67% accuracy despite advanced prompting techniques. + +Homepage: `https://osu-nlp-group.github.io/GroundCocoa/` + + +### Citation + +``` +@misc{kohli2025groundcocoabenchmarkevaluatingcompositional, + title={GroundCocoa: A Benchmark for Evaluating Compositional & Conditional Reasoning in Language Models}, + author={Harsh Kohli and Sachin Kumar and Huan Sun}, + year={2025}, + eprint={2404.04237}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2404.04237}, +} +``` + +### Groups and Tasks + +#### Groups + +- Not part of a group yet + +#### Tasks + +- `groundcocoa` + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/groundcocoa/groundcocoa.yaml b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/groundcocoa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..157c8e57eebfa3bab580f22113785e01ca5915d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/groundcocoa.yaml @@ -0,0 +1,18 @@ +task: groundcocoa +dataset_path: harsh147/GroundCocoa +output_type: multiple_choice +training_split: null +validation_split: validation +test_split: test +process_docs: !function utils.process_docs +doc_to_text: "{{criteria}}" +doc_to_target: gold +doc_to_choice: "choices" +target_delimiter: "" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +dataset_kwargs: + trust_remote_code: true + streaming: true diff --git a/lm-evaluation-harness/lm_eval/tasks/groundcocoa/utils.py b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..1d5509c412e8227af5c863888338942d8fca2bbc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/groundcocoa/utils.py @@ -0,0 +1,31 @@ +import datasets +import pandas as pd +from datasets import Dataset + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + cocoa_dataset = [sample for sample in dataset] + processed = [] + for doc in cocoa_dataset: + question = "A user has specified certain criteria for booking a flight. Below are five different flight options labeled 'A', 'B', 'C', 'D', and 'E'. Review these options and select the one that best matches the user requirements. Respond with a single option and the phrase 'The answer is Option ' followed by the correct letter - 'A', 'B', 'C', 'D', or 'E'\n\n" + question = question + "User Criteria: " + doc["query"] + question = question + "\n\n Option A: " + str(doc["Option A"]) + "\n" + question = question + "\n Option B: " + str(doc["Option B"]) + "\n" + question = question + "\n Option C: " + str(doc["Option C"]) + "\n" + question = question + "\n Option D: " + str(doc["Option D"]) + "\n" + question = question + "\n Option E: " + str(doc["Option E"]) + "\n" + out_doc = { + "criteria": question, + "choices": [ + "The answer is Option A", + "The answer is Option B", + "The answer is Option C", + "The answer is Option D", + "The answer is Option E", + ], + "gold": "The answer is Option " + doc["Answer"], + } + processed.append(out_doc) + df = pd.DataFrame(processed) + dataset = Dataset.from_pandas(df) + return dataset diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/README.md b/lm-evaluation-harness/lm_eval/tasks/gsm8k/README.md new file mode 100644 index 0000000000000000000000000000000000000000..1556151f821f526cf57388f15bb5c867af904a15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/README.md @@ -0,0 +1,62 @@ +# GSM8k + +## Paper +Training Verifiers to Solve Math Word Problems +https://arxiv.org/abs/2110.14168 + +State-of-the-art language models can match human performance on many tasks, but +they still struggle to robustly perform multi-step mathematical reasoning. To +diagnose the failures of current models and support research, we introduce GSM8K, +a dataset of 8.5K high quality linguistically diverse grade school math word problems. +We find that even the largest transformer models fail to achieve high test performance, +despite the conceptual simplicity of this problem distribution. + +NOTE: See the official implementation of the task: + https://github.com/openai/grade-school-math/blob/master/grade_school_math/calculator.py +for how to make use of the dataset's calculator annotations in your language +model's sample/generation function. + +Homepage: https://github.com/openai/grade-school-math + + +## Citation +``` +@misc{cobbe2021training, + title={Training Verifiers to Solve Math Word Problems}, + author={Karl Cobbe and Vineet Kosaraju and Mohammad Bavarian and Jacob Hilton and Reiichiro Nakano and Christopher Hesse and John Schulman}, + year={2021}, + eprint={2110.14168}, + archivePrefix={arXiv}, + primaryClass={cs.LG} +} +``` + +### Groups and Tasks + +#### Groups + +- `math_word_problems` +- `chain_of_thought` +- `self_consistency` + +#### Tasks + +- `gsm8k_yaml` +- `gsm8k_cot`: GSM8K with Chain-of-Thought +- `gsm8k_cot_self_consistency`: GSM8K with Chain-of-Thought and Self-Consistency +- `gsm8k_cot_llama`: GSM8K with prompt formatting modified to conform to the evaluation settings described by Meta here: https://huggingface.co/datasets/meta-llama/Meta-Llama-3.1-8B-Instruct-evals/viewer/Meta-Llama-3.1-8B-Instruct-evals__gsm8k__details?row=0 + - Use this task with --fewshot_as_multiturn and --apply_chat_template to replicate Meta's reported performance. + + +### Checklist + +- [x] Is in Eval-harness v1.0 ? +- [ ] Has been checked for regression from v1.0? +- [ ] Has been checked for equivalence with original paper methodology? +- [ ] "Main" checked variant clearly denoted? + +### Variant Wishlist + +- [ ] Variant with Calculator (see https://github.com/openai/grade-school-math/blob/master/grade_school_math/calculator.py for example implementation) +- [ ] Using Verifiers +- [ ] Majority voting "without CoT" diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-llama.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-llama.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e7948eeb8e3e7039f0c9c1738ac89aa19f4c4bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-llama.yaml @@ -0,0 +1,84 @@ +dataset_name: main +dataset_path: gsm8k +doc_to_target: '{{answer.split(''####'')[-1].strip() if answer is defined else target}}' +doc_to_text: "Given the following problem, reason and give a final answer to the problem.\nProblem: {{question}}\nYour response should end with \"The final answer is [answer]\" where [answer] is the response to the problem.\n" +fewshot_config: + sampler: first_n + samples: + - question: There are 15 trees in the grove. Grove workers will plant trees in the + grove today. After they are done, there will be 21 trees. How many trees did + the grove workers plant today? + target: There are 15 trees originally. Then there were 21 trees after some more + were planted. So there must have been 21 - 15 = 6. The final answer is 6 + - question: If there are 3 cars in the parking lot and 2 more cars arrive, how many + cars are in the parking lot? + target: There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The final answer + is 5 + - question: Leah had 32 chocolates and her sister had 42. If they ate 35, how many + pieces do they have left in total? + target: Originally, Leah had 32 chocolates. Her sister had 42. So in total they + had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The final answer is 39 + - question: Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 + lollipops. How many lollipops did Jason give to Denny? + target: Jason started with 20 lollipops. Then he had 12 after giving some to Denny. + So he gave Denny 20 - 12 = 8. The final answer is 8 + - question: Shawn has five toys. For Christmas, he got two toys each from his mom and + dad. How many toys does he have now? + target: Shawn started with 5 toys. If he got 2 toys each from his mom and dad, + then that is 4 more toys. 5 + 4 = 9. The final answer is 9 + - question: There were nine computers in the server room. Five more computers were + installed each day, from monday to thursday. How many computers are now in the + server room? + target: There were originally 9 computers. For each of 4 days, 5 more computers + were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The final answer is + 29 + - question: Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, + he lost 2 more. How many golf balls did he have at the end of wednesday? + target: Michael started with 58 golf balls. After losing 23 on tuesday, he had + 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The final answer + is 33 + - question: Olivia has $23. She bought five bagels for $3 each. How much money does + she have left? + target: Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 + dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The final answer is 8 +filter_list: +- filter: + - function: regex + group_select: -1 + regex_pattern: The final answer is ((-?[$0-9.,]{2,})|(-?[0-9]+)) + - function: take_first + name: strict-match +- filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +generation_kwargs: + do_sample: false + until: + - '<|eot_id|>' + - '<|start_header_id|>user<|end_header_id|>' + - 'Q:' + - + - <|im_end|> +tag: +- chain_of_thought +metadata: + version: 3.0 +metric_list: +- aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + metric: exact_match + regexes_to_ignore: + - ',' + - \$ + - '(?s).*#### ' + - \.$ +num_fewshot: 8 +output_type: generate_until +repeats: 1 +task: gsm8k_cot_llama +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-self-consistency.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-self-consistency.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0994081b049c0815ae85b9539b627e4c8df00dd3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-self-consistency.yaml @@ -0,0 +1,34 @@ +include: gsm8k-cot.yaml +tag: + - chain_of_thought + - self_consistency +task: gsm8k_cot_self_consistency +generation_kwargs: + until: + - "Q:" + - "\n\n" + do_sample: true + temperature: 0.2 +repeats: 64 +filter_list: + - name: "score-first" # pick only the first response, and report metrics on that + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "take_first" + - name: "maj@64" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "majority_vote" + - function: "take_first" + - name: "maj@8" # get Maj@8 , via selecting the first 8 responses. Using a better estimator would be optimal. + filter: + - function: "take_first_k" + k: 8 + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "majority_vote" + - function: "take_first" +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c112d324acf707e5934432068abd2ad6143438ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot-zeroshot.yaml @@ -0,0 +1,44 @@ +tag: + - math_word_problems +task: gsm8k_cot_zeroshot +dataset_path: gsm8k +dataset_name: main +output_type: generate_until +training_split: train +fewshot_split: train +test_split: test +doc_to_text: "Q: {{question}}\nA: Let's think step by step." +doc_to_target: "{{answer}}" #" {{answer.split('### ')[-1].rstrip()}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Q:" + - "" + - "<|im_end|>" + do_sample: false +repeats: 1 +num_fewshot: 0 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)." + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 3.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d125b0198535122fd5b12a388e903b03ee5f6020 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k-cot.yaml @@ -0,0 +1,83 @@ +dataset_name: main +dataset_path: gsm8k +doc_to_target: '{{answer.split(''####'')[-1].strip() if answer is defined else target}}' +doc_to_text: 'Q: {{question}} + + A:' +fewshot_config: + sampler: first_n + samples: + - question: There are 15 trees in the grove. Grove workers will plant trees in the + grove today. After they are done, there will be 21 trees. How many trees did + the grove workers plant today? + target: There are 15 trees originally. Then there were 21 trees after some more + were planted. So there must have been 21 - 15 = 6. The answer is 6. + - question: If there are 3 cars in the parking lot and 2 more cars arrive, how many + cars are in the parking lot? + target: There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The answer + is 5. + - question: Leah had 32 chocolates and her sister had 42. If they ate 35, how many + pieces do they have left in total? + target: Originally, Leah had 32 chocolates. Her sister had 42. So in total they + had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The answer is 39. + - question: Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 + lollipops. How many lollipops did Jason give to Denny? + target: Jason started with 20 lollipops. Then he had 12 after giving some to Denny. + So he gave Denny 20 - 12 = 8. The answer is 8. + - question: Shawn has five toys. For Christmas, he got two toys each from his mom and + dad. How many toys does he have now? + target: Shawn started with 5 toys. If he got 2 toys each from his mom and dad, + then that is 4 more toys. 5 + 4 = 9. The answer is 9. + - question: There were nine computers in the server room. Five more computers were + installed each day, from monday to thursday. How many computers are now in the + server room? + target: There were originally 9 computers. For each of 4 days, 5 more computers + were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The answer is + 29. + - question: Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, + he lost 2 more. How many golf balls did he have at the end of wednesday? + target: Michael started with 58 golf balls. After losing 23 on tuesday, he had + 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The answer + is 33. + - question: Olivia has $23. She bought five bagels for $3 each. How much money does + she have left? + target: Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 + dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The answer is 8. +filter_list: +- filter: + - function: regex + regex_pattern: The answer is (\-?[0-9\.\,]+). + - function: take_first + name: strict-match +- filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +generation_kwargs: + do_sample: false + until: + - 'Q:' + - + - <|im_end|> +tag: +- chain_of_thought +metadata: + version: 3.0 +metric_list: +- aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + metric: exact_match + regexes_to_ignore: + - ',' + - \$ + - '(?s).*#### ' + - \.$ +num_fewshot: 8 +output_type: generate_until +repeats: 1 +task: gsm8k_cot +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9d5bb39aedc0e2b991f0d79f2de6face47a31cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k/gsm8k.yaml @@ -0,0 +1,45 @@ +tag: + - math_word_problems +task: gsm8k +dataset_path: gsm8k +dataset_name: main +output_type: generate_until +training_split: train +fewshot_split: train +test_split: test +doc_to_text: "Question: {{question}}\nAnswer:" +doc_to_target: "{{answer}}" #" {{answer.split('### ')[-1].rstrip()}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Question:" + - "" + - "<|im_end|>" + do_sample: false + temperature: 0.0 +repeats: 1 +num_fewshot: 5 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "#### (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 3.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/README.md b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/README.md new file mode 100644 index 0000000000000000000000000000000000000000..932f7bb56c35bd76079379e4869ea12ad52df6bc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/README.md @@ -0,0 +1,54 @@ +# GSM8k Platinum + +GSM8K Platinum is a revised version of the full test set of GSM8K (Grade School Math 8K), a dataset of grade school math word problems. To revise this dataset, we ran a variety of frontier models each individual example and manually re-annotated any example for which at least one model made an error. We revise the labels of mislabeled examples, and remove any question that we determine to be poorly written (most often due to ambiguity in the problem statement). See our paper for further details on the revision process and our criteria for "bad" questions. + +## Paper +Do Large Language Model Benchmarks Test Reliability? +https://arxiv.org/abs/2502.03461 + +NOTE: See the official implementation of the task: + https://github.com/MadryLab/platinum-benchmarks/ +for how to make use of the dataset's calculator annotations in your language +model's sample/generation function. + +Blog: https://gradientscience.org/gsm8k-platinum/ +Homepage: http://platinum-bench.csail.mit.edu/ + + +## Citation +``` +@misc{vendrow2025largelanguagemodelbenchmarks, + title={Do Large Language Model Benchmarks Test Reliability?}, + author={Joshua Vendrow and Edward Vendrow and Sara Beery and Aleksander Madry}, + year={2025}, + eprint={2502.03461}, + archivePrefix={arXiv}, + primaryClass={cs.LG}, + url={https://arxiv.org/abs/2502.03461}, +} + +@misc{cobbe2021training, + title={Training Verifiers to Solve Math Word Problems}, + author={Karl Cobbe and Vineet Kosaraju and Mohammad Bavarian and Jacob Hilton and Reiichiro Nakano and Christopher Hesse and John Schulman}, + year={2021}, + eprint={2110.14168}, + archivePrefix={arXiv}, + primaryClass={cs.LG} +} +``` + +### Groups and Tasks + +#### Groups + +- `math_word_problems` +- `chain_of_thought` +- `self_consistency` + +#### Tasks + +- `gsm8k_platinum` +- `gsm8k_platinum_cot`: GSM8K Platinum with Chain-of-Thought +- `gsm8k_platinum_cot_self_consistency`: GSM8K Platinum with Chain-of-Thought and Self-Consistency +- `gsm8k_platinum_cot_llama`: GSM8K Platinum with prompt formatting modified to conform to the evaluation settings described by Meta here: https://huggingface.co/datasets/meta-llama/Meta-Llama-3.1-8B-Instruct-evals/viewer/Meta-Llama-3.1-8B-Instruct-evals__gsm8k__details?row=0 + - Use this task with --fewshot_as_multiturn and --apply_chat_template to replicate Meta's reported performance. diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-llama.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-llama.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e0dd54e6cc4708aedf96b42fa99176a855c8246 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-llama.yaml @@ -0,0 +1,84 @@ +task: gsm8k_platinum_cot_llama +dataset_name: main +dataset_path: madrylab/gsm8k-platinum +doc_to_target: '{{answer.split(''####'')[-1].strip() if answer is defined else target}}' +doc_to_text: "Given the following problem, reason and give a final answer to the problem.\nProblem: {{question}}\nYour response should end with \"The final answer is [answer]\" where [answer] is the response to the problem.\n" +fewshot_config: + sampler: first_n + samples: + - question: There are 15 trees in the grove. Grove workers will plant trees in the + grove today. After they are done, there will be 21 trees. How many trees did + the grove workers plant today? + target: There are 15 trees originally. Then there were 21 trees after some more + were planted. So there must have been 21 - 15 = 6. The final answer is 6 + - question: If there are 3 cars in the parking lot and 2 more cars arrive, how many + cars are in the parking lot? + target: There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The final answer + is 5 + - question: Leah had 32 chocolates and her sister had 42. If they ate 35, how many + pieces do they have left in total? + target: Originally, Leah had 32 chocolates. Her sister had 42. So in total they + had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The final answer is 39 + - question: Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 + lollipops. How many lollipops did Jason give to Denny? + target: Jason started with 20 lollipops. Then he had 12 after giving some to Denny. + So he gave Denny 20 - 12 = 8. The final answer is 8 + - question: Shawn has five toys. For Christmas, he got two toys each from his mom and + dad. How many toys does he have now? + target: Shawn started with 5 toys. If he got 2 toys each from his mom and dad, + then that is 4 more toys. 5 + 4 = 9. The final answer is 9 + - question: There were nine computers in the server room. Five more computers were + installed each day, from monday to thursday. How many computers are now in the + server room? + target: There were originally 9 computers. For each of 4 days, 5 more computers + were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The final answer is + 29 + - question: Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, + he lost 2 more. How many golf balls did he have at the end of wednesday? + target: Michael started with 58 golf balls. After losing 23 on tuesday, he had + 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The final answer + is 33 + - question: Olivia has $23. She bought five bagels for $3 each. How much money does + she have left? + target: Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 + dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The final answer is 8 +filter_list: +- filter: + - function: regex + group_select: -1 + regex_pattern: The final answer is ((-?[$0-9.,]{2,})|(-?[0-9]+)) + - function: take_first + name: strict-match +- filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +generation_kwargs: + do_sample: false + until: + - '<|eot_id|>' + - '<|start_header_id|>user<|end_header_id|>' + - 'Q:' + - + - <|im_end|> +tag: +- chain_of_thought +metadata: + version: 3.0 +metric_list: +- aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + metric: exact_match + regexes_to_ignore: + - ',' + - \$ + - '(?s).*#### ' + - \.$ +num_fewshot: 8 +output_type: generate_until +repeats: 1 +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-self-consistency.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-self-consistency.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2c60a0a346ae9f1589be0229082aa4a34ee5fc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-self-consistency.yaml @@ -0,0 +1,34 @@ +include: gsm8k-platinum-cot.yaml +tag: + - chain_of_thought + - self_consistency +task: gsm8k_platinum_cot_self_consistency +generation_kwargs: + until: + - "Q:" + - "\n\n" + do_sample: true + temperature: 0.2 +repeats: 64 +filter_list: + - name: "score-first" # pick only the first response, and report metrics on that + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "take_first" + - name: "maj@64" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "majority_vote" + - function: "take_first" + - name: "maj@8" # get Maj@8 , via selecting the first 8 responses. Using a better estimator would be optimal. + filter: + - function: "take_first_k" + k: 8 + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]*[0-9]+)" + - function: "majority_vote" + - function: "take_first" +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-zeroshot.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-zeroshot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f42a5f79ad94258135ba05c9d60fa39a05b6102f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot-zeroshot.yaml @@ -0,0 +1,44 @@ +tag: + - math_word_problems +task: gsm8k_platinum_cot_zeroshot +dataset_path: madrylab/gsm8k-platinum +dataset_name: main +output_type: generate_until +training_split: test +fewshot_split: test +test_split: test +doc_to_text: "Q: {{question}}\nA: Let's think step by step." +doc_to_target: "{{answer}}" #" {{answer.split('### ')[-1].rstrip()}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Q:" + - "" + - "<|im_end|>" + do_sample: false +repeats: 1 +num_fewshot: 0 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)." + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 3.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2790e49af3b5eb7824ef303dfaa1b528448e731f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum-cot.yaml @@ -0,0 +1,83 @@ +task: gsm8k_platinum_cot +dataset_name: main +dataset_path: madrylab/gsm8k-platinum +doc_to_target: '{{answer.split(''####'')[-1].strip() if answer is defined else target}}' +doc_to_text: 'Q: {{question}} + + A:' +fewshot_config: + sampler: first_n + samples: + - question: There are 15 trees in the grove. Grove workers will plant trees in the + grove today. After they are done, there will be 21 trees. How many trees did + the grove workers plant today? + target: There are 15 trees originally. Then there were 21 trees after some more + were planted. So there must have been 21 - 15 = 6. The answer is 6. + - question: If there are 3 cars in the parking lot and 2 more cars arrive, how many + cars are in the parking lot? + target: There are originally 3 cars. 2 more cars arrive. 3 + 2 = 5. The answer + is 5. + - question: Leah had 32 chocolates and her sister had 42. If they ate 35, how many + pieces do they have left in total? + target: Originally, Leah had 32 chocolates. Her sister had 42. So in total they + had 32 + 42 = 74. After eating 35, they had 74 - 35 = 39. The answer is 39. + - question: Jason had 20 lollipops. He gave Denny some lollipops. Now Jason has 12 + lollipops. How many lollipops did Jason give to Denny? + target: Jason started with 20 lollipops. Then he had 12 after giving some to Denny. + So he gave Denny 20 - 12 = 8. The answer is 8. + - question: Shawn has five toys. For Christmas, he got two toys each from his mom and + dad. How many toys does he have now? + target: Shawn started with 5 toys. If he got 2 toys each from his mom and dad, + then that is 4 more toys. 5 + 4 = 9. The answer is 9. + - question: There were nine computers in the server room. Five more computers were + installed each day, from monday to thursday. How many computers are now in the + server room? + target: There were originally 9 computers. For each of 4 days, 5 more computers + were added. So 5 * 4 = 20 computers were added. 9 + 20 is 29. The answer is + 29. + - question: Michael had 58 golf balls. On tuesday, he lost 23 golf balls. On wednesday, + he lost 2 more. How many golf balls did he have at the end of wednesday? + target: Michael started with 58 golf balls. After losing 23 on tuesday, he had + 58 - 23 = 35. After losing 2 more, he had 35 - 2 = 33 golf balls. The answer + is 33. + - question: Olivia has $23. She bought five bagels for $3 each. How much money does + she have left? + target: Olivia had 23 dollars. 5 bagels for 3 dollars each will be 5 x 3 = 15 + dollars. So she has 23 - 15 dollars left. 23 - 15 is 8. The answer is 8. +filter_list: +- filter: + - function: regex + regex_pattern: The answer is (\-?[0-9\.\,]+). + - function: take_first + name: strict-match +- filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +generation_kwargs: + do_sample: false + until: + - 'Q:' + - + - <|im_end|> +tag: +- chain_of_thought +metadata: + version: 3.0 +metric_list: +- aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + metric: exact_match + regexes_to_ignore: + - ',' + - \$ + - '(?s).*#### ' + - \.$ +num_fewshot: 8 +output_type: generate_until +repeats: 1 +test_split: test diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14502f66e0aa7b81afe80e11bbb63570234e7913 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm8k_platinum/gsm8k-platinum.yaml @@ -0,0 +1,45 @@ +tag: + - math_word_problems +task: gsm8k_platinum +dataset_path: madrylab/gsm8k-platinum +dataset_name: main +output_type: generate_until +training_split: test +fewshot_split: test +test_split: test +doc_to_text: "Question: {{question}}\nAnswer:" +doc_to_target: "{{answer}}" #" {{answer.split('### ')[-1].rstrip()}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Question:" + - "" + - "<|im_end|>" + do_sample: false + temperature: 0.0 +repeats: 1 +num_fewshot: 5 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "#### (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 3.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm_plus/README.md b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/README.md new file mode 100644 index 0000000000000000000000000000000000000000..173a8a5e8225c9c69314d93241c4304802b54bc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/README.md @@ -0,0 +1,48 @@ +# gsm_plus + +### Paper + +Title: `GSM-PLUS: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers` + +Abstract: `Large language models (LLMs) have achieved impressive performance across various mathematical reasoning benchmarks. However, there are increasing debates regarding whether these models truly understand and apply mathematical knowledge or merely rely on shortcuts for mathematical reasoning. One essential and frequently occurring evidence is that when the math questions are slightly changed, LLMs can behave incorrectly. This motivates us to evaluate the robustness of LLMs’ math reasoning capability by testing a wide range of question variations. We introduce the adversarial grade school math (GSM-PLUS) dataset, an extension of GSM8K augmented with various mathematical perturbations. Our experiments on 25 LLMs and 4 prompting techniques show that while LLMs exhibit different levels of math reasoning abilities, their performances are far from robust. In particular, even for problems that have been solved in GSM8K, LLMs can make mistakes when new statements are added or the question targets are altered. We also explore whether more robust performance can be achieved by composing existing prompting methods, in which we try an iterative method that generates and verifies each intermediate thought based on its reasoning goal and calculation result.` + +Homepage: https://huggingface.co/datasets/qintongli/GSM-Plus + +### Citation + +```bibtex +@misc{li2024gsmpluscomprehensivebenchmarkevaluating, + title={GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers}, + author={Qintong Li and Leyang Cui and Xueliang Zhao and Lingpeng Kong and Wei Bi}, + year={2024}, + eprint={2402.19255}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2402.19255}, +} +``` + +### Groups and Tasks + +#### Groups + +* Not part of a group yet + +#### Tasks + +The following tasks evaluate subjects in the gsm_plus dataset +- `gsm_plus` +- `gsm_plus_mini` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eb08f1ad8a543dc07d89ee02aed7d9e986c844b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus.yaml @@ -0,0 +1,44 @@ +tag: + - math_word_problems +task: gsm_plus +dataset_path: qintongli/GSM-Plus +output_type: generate_until +training_split: test +fewshot_split: test +test_split: test +doc_to_text: "Question: {{question}}\nAnswer:" +doc_to_target: "{{solution}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Question:" + - "" + - "<|im_end|>" + do_sample: false + temperature: 0.0 +repeats: 1 +num_fewshot: 5 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "#### (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus_mini.yaml b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus_mini.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b03cfd2ea8dd639c1fc2135fc288cd2f7367cd9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/gsm_plus/gsm_plus_mini.yaml @@ -0,0 +1,44 @@ +tag: + - math_word_problems +task: gsm_plus_mini +dataset_path: qintongli/GSM-Plus +output_type: generate_until +training_split: testmini +fewshot_split: testmini +test_split: testmini +doc_to_text: "Question: {{question}}\nAnswer:" +doc_to_target: "{{solution}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: false + regexes_to_ignore: + - "," + - "\\$" + - "(?s).*#### " + - "\\.$" +generation_kwargs: + until: + - "Question:" + - "" + - "<|im_end|>" + do_sample: false + temperature: 0.0 +repeats: 1 +num_fewshot: 5 +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "#### (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - name: "flexible-extract" + filter: + - function: "regex" + group_select: -1 + regex_pattern: "(-?[$0-9.,]{2,})|(-?[0-9]+)" + - function: "take_first" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/README.md b/lm-evaluation-harness/lm_eval/tasks/haerae/README.md new file mode 100644 index 0000000000000000000000000000000000000000..108626ae34ba4deb88d22b2ca02f43c54d2fcb5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/README.md @@ -0,0 +1,49 @@ +# HAE-RAE BENCH + +### Paper + +Title: `HAE-RAE Bench: Evaluation of Korean Knowledge in Language Models` + +Abstract: `Large Language Models (LLMs) trained on massive corpora demonstrate impressive capabilities in a wide range of tasks. While there are ongoing efforts to adapt these models to languages beyond English, the attention given to their evaluation methodologies remains limited. Current multilingual benchmarks often rely on back translations or re-implementations of English tests, limiting their capacity to capture unique cultural and linguistic nuances. To bridge this gap for the Korean language, we introduce HAE-RAE Bench, a dataset curated to challenge models lacking Korean cultural and contextual depth. The dataset encompasses six downstream tasks across four domains: vocabulary, history, general knowledge, and reading comprehension. Contrary to traditional evaluation suites focused on token or sequence classification and specific mathematical or logical reasoning, HAE-RAE Bench emphasizes a model's aptitude for recalling Korean-specific knowledge and cultural contexts. Comparative analysis with prior Korean benchmarks indicates that the HAE-RAE Bench presents a greater challenge to non-native models, by disturbing abilities and knowledge learned from English being transferred.` + +Homepage: https://huggingface.co/datasets/HAERAE-HUB/HAE_RAE_BENCH + +### Citation + +@misc{son2023haerae, + title={HAE-RAE Bench: Evaluation of Korean Knowledge in Language Models}, + author={Guijin Son and Hanwool Lee and Suwan Kim and Huiseo Kim and Jaecheol Lee and Je Won Yeom and Jihyu Jung and Jung Woo Kim and Songseong Kim}, + year={2023}, + eprint={2309.02706}, + archivePrefix={arXiv}, + primaryClass={cs.CL} +} + +### Groups and Tasks + +#### Groups + +* `haerae`: 'It consists of five tasks provided in the HAERAE-BENCH paper. 'Reading Comprehension' was excluded from the implementation due to copyright issues. We will include it in the next haerae update. For other tasks, some part of data may be replaced or increased with the production of Haerae v1.1. Please note this when using it.' + +#### Tasks + +The following tasks evaluate subjects in the HaeRae dataset + +- `haerae_standard_nomenclature` +- `haerae_loan_word` +- `haerae_rare_word` +- `haerae_general_knowledge` +- `haerae_history` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/_default_haerae_yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/_default_haerae_yaml new file mode 100644 index 0000000000000000000000000000000000000000..807c10e0850078de24227d2738093e4511079690 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/_default_haerae_yaml @@ -0,0 +1,16 @@ +dataset_path: HAERAE-HUB/HAE_RAE_BENCH +test_split: test +fewshot_split: test +output_type: multiple_choice +doc_to_text: "{{query}}" +doc_to_choice: ["(A)", "(B)", "(C)", "(D)", "(E)"] +doc_to_target: "{{answer}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/_haerae.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/_haerae.yaml new file mode 100644 index 0000000000000000000000000000000000000000..acf413fb8293c5f32001305e09d25dfd6b1dfc5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/_haerae.yaml @@ -0,0 +1,16 @@ +group: haerae +task: + - haerae_general_knowledge + - haerae_history + - haerae_loan_word + - haerae_rare_word + - haerae_standard_nomenclature +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_gk.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_gk.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97d2bd71d8b9a333a16a3065a14276e0b49926da --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_gk.yaml @@ -0,0 +1,3 @@ +dataset_name: general_knowledge +include: _default_haerae_yaml +task: haerae_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_hi.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_hi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed366912762cf7d1784fdef9103223fc82ab4c70 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_hi.yaml @@ -0,0 +1,3 @@ +dataset_name: history +include: _default_haerae_yaml +task: haerae_history diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_lw.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_lw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1cb572784f06fa9b1299d8c2c817bc9541ce646b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_lw.yaml @@ -0,0 +1,3 @@ +dataset_name: loan_words +include: _default_haerae_yaml +task: haerae_loan_word diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_rw.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_rw.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cfa2d1cf472e0b5a58231d6519f8e62bb8295c5c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_rw.yaml @@ -0,0 +1,3 @@ +dataset_name: rare_words +include: _default_haerae_yaml +task: haerae_rare_word diff --git a/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_sn.yaml b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_sn.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66bf43e060e7af557274653ccbe626194f91c94e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/haerae/haerae_sn.yaml @@ -0,0 +1,3 @@ +dataset_name: standard_nomenclature +include: _default_haerae_yaml +task: haerae_standard_nomenclature diff --git a/lm-evaluation-harness/lm_eval/tasks/headqa/README.md b/lm-evaluation-harness/lm_eval/tasks/headqa/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9e061f0ed44e65ef04cc9d98220058051d509da6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/headqa/README.md @@ -0,0 +1,57 @@ +# HEAD-QA + +### Paper + +HEAD-QA: A Healthcare Dataset for Complex Reasoning +https://arxiv.org/pdf/1906.04701.pdf + +HEAD-QA is a multi-choice HEAlthcare Dataset. The questions come from exams to access a specialized position in the +Spanish healthcare system, and are challenging even for highly specialized humans. They are designed by the Ministerio +de Sanidad, Consumo y Bienestar Social. +The dataset contains questions about the following topics: medicine, nursing, psychology, chemistry, pharmacology and biology. + +Homepage: https://aghie.github.io/head-qa/ + + +### Citation + +``` +@inproceedings{vilares-gomez-rodriguez-2019-head, + title = "{HEAD}-{QA}: A Healthcare Dataset for Complex Reasoning", + author = "Vilares, David and + G{\'o}mez-Rodr{\'i}guez, Carlos", + booktitle = "Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics", + month = jul, + year = "2019", + address = "Florence, Italy", + publisher = "Association for Computational Linguistics", + url = "https://www.aclweb.org/anthology/P19-1092", + doi = "10.18653/v1/P19-1092", + pages = "960--966", + abstract = "We present HEAD-QA, a multi-choice question answering testbed to encourage research on complex reasoning. The questions come from exams to access a specialized position in the Spanish healthcare system, and are challenging even for highly specialized humans. We then consider monolingual (Spanish) and cross-lingual (to English) experiments with information retrieval and neural techniques. We show that: (i) HEAD-QA challenges current methods, and (ii) the results lag well behind human performance, demonstrating its usefulness as a benchmark for future work.", +} +``` + +### Groups and Tasks + +#### Groups + +- `headqa`: Evaluates `headqa_en` and `headqa_es` + +#### Tasks + +* `headqa_en` - English variant of HEAD-QA +* `headqa_es` - Spanish variant of HEAD-QA + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant?\ + * [x] Same as LM Evaluation Harness v0.3.0 implementation diff --git a/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_en.yaml b/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d1b0b3e015240cc3285ffe67c167901bfa385e1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_en.yaml @@ -0,0 +1,22 @@ +tag: headqa +task: headqa_en +dataset_path: EleutherAI/headqa +dataset_name: en +output_type: multiple_choice +training_split: train +validation_split: validation +test_split: test +doc_to_text: "Question: {{qtext}}\nAnswer:" +doc_to_target: "{{ra - 1}}" +doc_to_choice: "{{answers|map(attribute='atext')|list}}" # this will be cast to an int. +should_decontaminate: true +doc_to_decontamination_query: query +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_es.yaml b/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_es.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88e202f753e18f6fd6b8e303353cc0f38fce73e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/headqa/headqa_es.yaml @@ -0,0 +1,3 @@ +include: headqa_en.yaml +task: headqa_es +dataset_name: es diff --git a/lm-evaluation-harness/lm_eval/tasks/hellaswag/README.md b/lm-evaluation-harness/lm_eval/tasks/hellaswag/README.md new file mode 100644 index 0000000000000000000000000000000000000000..9fdbac13581c06430b63248514b7cf5c9610c220 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hellaswag/README.md @@ -0,0 +1,49 @@ +# HellaSwag + +### Paper + +Title: `HellaSwag: Can a Machine Really Finish Your Sentence?` + +Abstract: https://arxiv.org/abs/1905.07830 + +Recent work by Zellers et al. (2018) introduced a new task of commonsense natural language inference: given an event description such as "A woman sits at a piano," a machine must select the most likely followup: "She sets her fingers on the keys." With the introduction of BERT, near human-level performance was reached. Does this mean that machines can perform human level commonsense inference? +In this paper, we show that commonsense inference still proves difficult for even state-of-the-art models, by presenting HellaSwag, a new challenge dataset. Though its questions are trivial for humans (>95% accuracy), state-of-the-art models struggle (<48%). We achieve this via Adversarial Filtering (AF), a data collection paradigm wherein a series of discriminators iteratively select an adversarial set of machine-generated wrong answers. AF proves to be surprisingly robust. The key insight is to scale up the length and complexity of the dataset examples towards a critical 'Goldilocks' zone wherein generated text is ridiculous to humans, yet often misclassified by state-of-the-art models. +Our construction of HellaSwag, and its resulting difficulty, sheds light on the inner workings of deep pretrained models. More broadly, it suggests a new path forward for NLP research, in which benchmarks co-evolve with the evolving state-of-the-art in an adversarial way, so as to present ever-harder challenges. + +Homepage: `https://rowanzellers.com/hellaswag/` + + +### Citation + +``` +@inproceedings{zellers2019hellaswag, + title={HellaSwag: Can a Machine Really Finish Your Sentence?}, + author={Zellers, Rowan and Holtzman, Ari and Bisk, Yonatan and Farhadi, Ali and Choi, Yejin}, + booktitle ={Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics}, + year={2019} +} +``` + +### Groups and Tasks + +#### Groups + +- Not part of a group yet + +#### Tasks + +- `hellaswag` + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-310.pyc b/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..551eab531702021d5c6001d4aed88db3f7742efc Binary files /dev/null and b/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-310.pyc differ diff --git a/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-311.pyc b/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8150eff5e350e9ee5fbea4a8f5ea8866442fe232 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/tasks/hellaswag/__pycache__/utils.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/tasks/hellaswag/hellaswag.yaml b/lm-evaluation-harness/lm_eval/tasks/hellaswag/hellaswag.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0735bbd97773488d99c047ab1342ad65aca4142 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hellaswag/hellaswag.yaml @@ -0,0 +1,24 @@ +tag: + - multiple_choice +task: hellaswag +dataset_path: hellaswag +dataset_name: null +output_type: multiple_choice +training_split: train +validation_split: validation +test_split: null +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{label}}" +doc_to_choice: "choices" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/hellaswag/utils.py b/lm-evaluation-harness/lm_eval/tasks/hellaswag/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..b526a9e93076f7db54221072d58ca4bd7161ee97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hellaswag/utils.py @@ -0,0 +1,25 @@ +import re + +import datasets + + +def preprocess(text): + text = text.strip() + # NOTE: Brackets are artifacts of the WikiHow dataset portion of HellaSwag. + text = text.replace(" [title]", ". ") + text = re.sub("\\[.*?\\]", "", text) + text = text.replace(" ", " ") + return text + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + ctx = doc["ctx_a"] + " " + doc["ctx_b"].capitalize() + out_doc = { + "query": preprocess(doc["activity_label"] + ": " + ctx), + "choices": [preprocess(ending) for ending in doc["endings"]], + "gold": int(doc["label"]), + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/README.md b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/README.md new file mode 100644 index 0000000000000000000000000000000000000000..ce98279e3eafd134d72658f3db0c9af5eaf755e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/README.md @@ -0,0 +1,54 @@ +# ETHICS Dataset + +### Paper + +Pointer Sentinel Mixture Models +https://arxiv.org/pdf/1609.07843.pdf + +The ETHICS dataset is a benchmark that spans concepts in justice, well-being, +duties, virtues, and commonsense morality. Models predict widespread moral +judgments about diverse text scenarios. This requires connecting physical and +social world knowledge to value judgements, a capability that may enable us +to steer chatbot outputs or eventually regularize open-ended reinforcement +learning agents. + +Homepage: https://github.com/hendrycks/ethics + +### Citation + +``` +@article{hendrycks2021ethics + title={Aligning AI With Shared Human Values}, + author={Dan Hendrycks and Collin Burns and Steven Basart and Andrew Critch and Jerry Li and Dawn Song and Jacob Steinhardt}, + journal={Proceedings of the International Conference on Learning Representations (ICLR)}, + year={2021} +} +``` + +### Groups and Tasks + +#### Groups + +- `hendrycks_ethics` + +#### Tasks + +* `ethics_cm` +* `ethics_deontology` +* `ethics_justice` +* `ethics_utilitarianism` +* (MISSING) `ethics_utilitarianism_original` +* `ethics_virtue` + +### Checklist + +* [x] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [x] Is the "Main" variant of this task clearly denoted? +* [x] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? + * [ ] Matches v0.3.0 of Eval Harness diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/commonsense.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/commonsense.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4f134375c24617215121bda5e43a98f8b88e1b2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/commonsense.yaml @@ -0,0 +1,15 @@ +tag: + - hendrycks_ethics +task: ethics_cm +dataset_path: EleutherAI/hendrycks_ethics +dataset_name: commonsense +output_type: multiple_choice +training_split: train +test_split: test +doc_to_text: "{{input}}\nQuestion: Is this wrong?\nAnswer:" +doc_to_target: label +doc_to_choice: ['no', 'yes'] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/deontology.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/deontology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2f36852226e86a817023cf428b00dc3b93ac468 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/deontology.yaml @@ -0,0 +1,9 @@ +include: commonsense.yaml +task: ethics_deontology +dataset_name: deontology +doc_to_text: "Question: Would most people believe this reasonable or unreasonable to say? \"{{scenario}} {{excuse.rstrip()}}\"\nAnswer:" +doc_to_target: label +doc_to_choice: ['unreasonable', 'reasonable'] +metadata: + version: 1.0 +# TODO: implement exact-match metric for this subset diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/justice.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/justice.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a67f55c708f4991fb67935787aa9a952a32178cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/justice.yaml @@ -0,0 +1,9 @@ +include: deontology.yaml +tag: + - hendrycks_ethics +task: ethics_justice +dataset_name: justice +doc_to_text: "Question: Would most people believe this reasonable or unreasonable to say? \"{{scenario}}\"\nAnswer:" +# TODO: impl. exact match for this and deontology +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81134de900dc37d571db2ea53a4746022c8bceac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism.yaml @@ -0,0 +1,12 @@ +include: commonsense.yaml +tag: + - hendrycks_ethics +task: ethics_utilitarianism +dataset_name: utilitarianism +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +doc_to_choice: ['no', 'yes'] +metric_list: + - metric: acc +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism_original_yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism_original_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5583c1b6829e6f0b25093e5b944ea5ca048a0764 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utilitarianism_original_yaml @@ -0,0 +1,16 @@ +# group: +# - hendrycks_ethics +# task: ethics_utilitarianism_original +# dataset_path: hails/hendrycks_ethics +# dataset_name: utilitarianism +# output_type: winograd_schema +# fewshot_split: null # TODO: implement a special fewshot split for this dataset subsets +# test_split: test +# template_aliases: #"{% set answer_choices = range(1, 11)|list %}" +# doc_to_text: 'Activity: "{{activity}}"\nRating:' +# doc_to_target: "{{answer_choices[label]}}" +# metric_list: +# - metric: acc +# TODO: we want this to be implemented as a winograd_schema task type, actually +# metadata: +# version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utils.py b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..1ff0daa961c20daaa5dde14fe73d464277c1750a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/utils.py @@ -0,0 +1,25 @@ +import random + + +### Utils for `ethics_utilitarianism` task below +def _preproc_doc(doc): + rnd = random.Random(doc["activity"]) + scenarios = [doc["activity"], doc["baseline"]] + ordering = [0, 1] + rnd.shuffle(ordering) + doc = { + "scenarios": [scenarios[ordering[0]], scenarios[ordering[1]]], + # The correct scenario is always first + "label": int(ordering.index(0) == 0), + } + return doc + + +def doc_to_text(doc) -> str: + doc = _preproc_doc(doc) + return f"Scenario 1: {doc['scenarios'][0]}\nScenario 2: {doc['scenarios'][1]}\nQuestion: Is Scenario 1 preferable?\nAnswer:" + + +def doc_to_target(doc): + doc = _preproc_doc(doc) + return doc["label"] diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/virtue.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/virtue.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b456e4a5a49a4f3cd626e4cca48154032e08367f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_ethics/virtue.yaml @@ -0,0 +1,10 @@ +include: commonsense.yaml +tag: + - hendrycks_ethics +task: ethics_virtue +dataset_name: virtue +doc_to_text: "Sentence: {{scenario}}\nQuestion: Does the character in this sentence exhibit the trait \"{{trait}}\"?\nAnswer:" +doc_to_target: label +doc_to_choice: ['no', 'yes'] +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d01cf9b2465b4e825ed8b5c67fe0aca281c31781 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math.yaml @@ -0,0 +1,15 @@ +group: hendrycks_math +task: + - hendrycks_math_algebra + - hendrycks_math_counting_and_prob + - hendrycks_math_geometry + - hendrycks_math_intermediate_algebra + - hendrycks_math_num_theory + - hendrycks_math_prealgebra + - hendrycks_math_precalc +aggregate_metric_list: + - metric: exact_match + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_algebra.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_algebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9ce9c9c5ad64d894732054caa679d73b1df23881 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_algebra.yaml @@ -0,0 +1,25 @@ +tag: + - math_word_problems +task: hendrycks_math_algebra +dataset_path: EleutherAI/hendrycks_math +process_docs: !function utils.process_docs +dataset_name: algebra +output_type: generate_until +training_split: train +test_split: test +doc_to_text: "Problem: {{problem}}\nAnswer:" +process_results: !function utils.process_results +doc_to_target: "{{answer}}" +generation_kwargs: + until: + - "Problem:" + do_sample: false + temperature: 0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 +dataset_kwargs: + trust_remote_code: true diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_counting_and_prob.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_counting_and_prob.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6de3e140c252803afd1246e155c14c9c66672351 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_counting_and_prob.yaml @@ -0,0 +1,3 @@ +include: hendrycks_math_algebra.yaml +dataset_name: counting_and_probability +task: hendrycks_math_counting_and_prob diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_geometry.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_geometry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2016439a8673cd970fe8869b61f001ee71c54b36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_geometry.yaml @@ -0,0 +1,3 @@ +include: hendrycks_math_algebra.yaml +dataset_name: geometry +task: hendrycks_math_geometry diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_num_theory.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_num_theory.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a1e53bc306336834dde059168d35fefc454d0b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_num_theory.yaml @@ -0,0 +1,3 @@ +include: hendrycks_math_algebra.yaml +dataset_name: number_theory +task: hendrycks_math_num_theory diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_prealgebra.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_prealgebra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a0baa388a3531cbe00765fb545b83ae2b540842 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_prealgebra.yaml @@ -0,0 +1,3 @@ +include: hendrycks_math_algebra.yaml +dataset_name: prealgebra +task: hendrycks_math_prealgebra diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_precalc.yaml b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_precalc.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af81fec9362204da67b8bb4f193773a0cc67cb67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/hendrycks_math_precalc.yaml @@ -0,0 +1,3 @@ +include: hendrycks_math_algebra.yaml +dataset_name: precalculus +task: hendrycks_math_precalc diff --git a/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/utils.py b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..0edd59a16a30be95f79403239e73af65c12d8d66 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hendrycks_math/utils.py @@ -0,0 +1,231 @@ +from typing import Dict, List + +import datasets + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc: dict) -> dict: + out_doc = { + "problem": doc["problem"], + "solution": doc["solution"], + "answer": remove_boxed(last_boxed_only_string(doc["solution"])), + } + return out_doc + + return dataset.map(_process_doc) + + +def process_results(doc: dict, results: List[str]) -> Dict[str, int]: + retval = 0 + indices = [pos for pos, char in enumerate(results[0]) if char == "$"] + if len(indices) <= 1: + answer = results[0] + else: + answer = results[0][indices[0] + 1 : indices[-1]] + + if is_equiv(answer, remove_boxed(last_boxed_only_string(doc["solution"]))): + retval = 1 + + results = { + "exact_match": retval, + } + return results + + +# string normalization from https://github.com/EleutherAI/lm-evaluation-harness/blob/master/lm_eval/tasks/hendrycks_math.py +def is_equiv(str1, str2, verbose=False): + if str1 is None and str2 is None: + print("WARNING: Both None") + return True + if str1 is None or str2 is None: + return False + + try: + ss1 = strip_string(str1) + ss2 = strip_string(str2) + if verbose: + print(ss1, ss2) + return ss1 == ss2 + except Exception: + return str1 == str2 + + +def remove_boxed(s): + if "\\boxed " in s: + left = "\\boxed " + assert s[: len(left)] == left + return s[len(left) :] + + left = "\\boxed{" + + assert s[: len(left)] == left + assert s[-1] == "}" + + return s[len(left) : -1] + + +def last_boxed_only_string(string): + idx = string.rfind("\\boxed") + if "\\boxed " in string: + return "\\boxed " + string.split("\\boxed ")[-1].split("$")[0] + if idx < 0: + idx = string.rfind("\\fbox") + if idx < 0: + return None + + i = idx + right_brace_idx = None + num_left_braces_open = 0 + while i < len(string): + if string[i] == "{": + num_left_braces_open += 1 + if string[i] == "}": + num_left_braces_open -= 1 + if num_left_braces_open == 0: + right_brace_idx = i + break + i += 1 + + if right_brace_idx is None: + retval = None + else: + retval = string[idx : right_brace_idx + 1] + + return retval + + +def fix_fracs(string): + substrs = string.split("\\frac") + new_str = substrs[0] + if len(substrs) > 1: + substrs = substrs[1:] + for substr in substrs: + new_str += "\\frac" + if substr[0] == "{": + new_str += substr + else: + try: + assert len(substr) >= 2 + except AssertionError: + return string + a = substr[0] + b = substr[1] + if b != "{": + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}{" + b + "}" + post_substr + else: + new_str += "{" + a + "}{" + b + "}" + else: + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}" + b + post_substr + else: + new_str += "{" + a + "}" + b + string = new_str + return string + + +def fix_a_slash_b(string): + if len(string.split("/")) != 2: + return string + a = string.split("/")[0] + b = string.split("/")[1] + try: + a = int(a) + b = int(b) + assert string == "{}/{}".format(a, b) + new_string = "\\frac{" + str(a) + "}{" + str(b) + "}" + return new_string + except AssertionError: + return string + + +def remove_right_units(string): + # "\\text{ " only ever occurs (at least in the val set) when describing units + if "\\text{ " in string: + splits = string.split("\\text{ ") + assert len(splits) == 2 + return splits[0] + else: + return string + + +def fix_sqrt(string): + if "\\sqrt" not in string: + return string + splits = string.split("\\sqrt") + new_string = splits[0] + for split in splits[1:]: + if split[0] != "{": + a = split[0] + new_substr = "\\sqrt{" + a + "}" + split[1:] + else: + new_substr = "\\sqrt" + split + new_string += new_substr + return new_string + + +def strip_string(string): + # linebreaks + string = string.replace("\n", "") + + # remove inverse spaces + string = string.replace("\\!", "") + + # replace \\ with \ + string = string.replace("\\\\", "\\") + + # replace tfrac and dfrac with frac + string = string.replace("tfrac", "frac") + string = string.replace("dfrac", "frac") + + # remove \left and \right + string = string.replace("\\left", "") + string = string.replace("\\right", "") + + # Remove circ (degrees) + string = string.replace("^{\\circ}", "") + string = string.replace("^\\circ", "") + + # remove dollar signs + string = string.replace("\\$", "") + + # remove units (on the right) + string = remove_right_units(string) + + # remove percentage + string = string.replace("\\%", "") + string = string.replace("\%", "") # noqa: W605 + + # " 0." equivalent to " ." and "{0." equivalent to "{." Alternatively, add "0" if "." is the start of the string + string = string.replace(" .", " 0.") + string = string.replace("{.", "{0.") + # if empty, return empty string + if len(string) == 0: + return string + if string[0] == ".": + string = "0" + string + + # to consider: get rid of e.g. "k = " or "q = " at beginning + if len(string.split("=")) == 2: + if len(string.split("=")[0]) <= 2: + string = string.split("=")[1] + + # fix sqrt3 --> sqrt{3} + string = fix_sqrt(string) + + # remove spaces + string = string.replace(" ", "") + + # \frac1b or \frac12 --> \frac{1}{b} and \frac{1}{2}, etc. Even works with \frac1{72} (but not \frac{72}1). Also does a/b --> \\frac{a}{b} + string = fix_fracs(string) + + # manually change 0.5 --> \frac{1}{2} + if string == "0.5": + string = "\\frac{1}{2}" + + # NOTE: X/Y changed to \frac{X}{Y} in dataset, but in simple cases fix in case the model output is X/Y + string = fix_a_slash_b(string) + + return string diff --git a/lm-evaluation-harness/lm_eval/tasks/histoires_morales/histoires_morales.yaml b/lm-evaluation-harness/lm_eval/tasks/histoires_morales/histoires_morales.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88fcc402f6142251b5ce630b9610956082b0f1dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/histoires_morales/histoires_morales.yaml @@ -0,0 +1,17 @@ +task: histoires_morales +dataset_path: LabHC/histoires_morales +output_type: multiple_choice +test_split: train +process_docs: !function utils.process_docs +doc_to_text: "{{query}}" +doc_to_target: "{{label}}" +doc_to_choice: "choices" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/histoires_morales/utils.py b/lm-evaluation-harness/lm_eval/tasks/histoires_morales/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..2e996b7466a863178daac352ae6a892bd934def5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/histoires_morales/utils.py @@ -0,0 +1,21 @@ +import datasets + + +def process_docs(dataset: datasets.Dataset) -> datasets.Dataset: + def _process_doc(doc): + ctx = ( + doc["norm"].capitalize() + + " " + + doc["situation"].capitalize() + + " " + + doc["intention"].capitalize() + ) + choices = [doc["moral_action"], doc["immoral_action"]] + out_doc = { + "query": ctx, + "choices": choices, + "label": 0, + } + return out_doc + + return dataset.map(_process_doc) diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/_hrm8k_yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/_hrm8k_yaml new file mode 100644 index 0000000000000000000000000000000000000000..18c53d22997061ed43854d449e5b9e64a80e7335 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/_hrm8k_yaml @@ -0,0 +1,22 @@ +dataset_path: HAERAE-HUB/HRM8K +output_type: generate_until +test_split: test +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +process_results: !function utils.process_results +num_fewshot: 0 +generation_kwargs: + until: + - "" + - "<|end_of_text|>" + - "<|endoftext|>" + - "<|im_end|>" + max_gen_toks: 512 + do_sample: false + temperature: 0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc9753f63beb50d2f041245093a186a9b64cd7ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k.yaml @@ -0,0 +1,13 @@ +group: hrm8k +task: + - hrm8k_gsm8k + - hrm8k_ksm + - hrm8k_math + - hrm8k_mmmlu + - hrm8k_omni_math +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_gsm8k.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_gsm8k.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a46ff5a04378c03d144c225ad164bd6f9b9cb1c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_gsm8k.yaml @@ -0,0 +1,3 @@ +include: _hrm8k_yaml +dataset_name: GSM8K +task: hrm8k_gsm8k diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_ksm.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_ksm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c1f7ac230e402c8a59bf85073b27ec2bf5722c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_ksm.yaml @@ -0,0 +1,3 @@ +include: _hrm8k_yaml +dataset_name: KSM +task: hrm8k_ksm diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_mmmlu.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_mmmlu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20faaaf10f1374df40f6cd0fe57ab8c330909c7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_mmmlu.yaml @@ -0,0 +1,4 @@ +include: _hrm8k_yaml +dataset_name: MMMLU +task: hrm8k_mmmlu +doc_to_text: !function utils.doc_to_text_mmmlu diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_omni_math.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_omni_math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2dadac2dbdd5aab1e2c5530b0337677f2e92d7e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/hrm8k_omni_math.yaml @@ -0,0 +1,3 @@ +include: _hrm8k_yaml +dataset_name: OMNI_MATH +task: hrm8k_omni_math diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/utils.py b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..aaeecd1495fd83dc93af64f92f2b5dd4f3277224 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/default/utils.py @@ -0,0 +1,285 @@ +import re +from typing import Dict, List + + +def doc_to_text(doc): + text = ( + "주어진 문제를 풀어보세요.\n" + "문제를 푼 후, 최종 답변을 다음과 같은 형식으로 작성하세요: $\\boxed{N}$.\n\n" + f"문제: {doc['question'].strip()}\n답변:" + ) + return text + + +def doc_to_text_mmmlu(doc): + text = ( + "주어진 문제를 풀어보세요.\n" + "문제를 푼 후, 주어진 선택지 (1, 2, 3, 4) 중 최종 선택지를 다음 형식으로 작성하세요: $\\boxed{N}$.\n\n" + f"문제: {doc['question'].strip()}\n답변:" + ) + return text + + +def doc_to_target(doc): + return postprocess(doc["answer"]) + + +def postprocess(s): + s = str(s).strip() + try: + float_value = float(s) + return str(int(float_value)) if float_value.is_integer() else str(float_value) + except Exception: + return s + + +def process_results(doc: dict, results: List[str]) -> Dict[str, int]: + candidate = results[0] + + gold = postprocess(doc["answer"]) + + if not gold: + print(doc, candidate, gold) + if is_equiv(candidate, gold): + retval = 1 + else: + retval = 0 + + results = { + "exact_match": retval, + } + return results + + +def is_equiv(str1, str2, verbose=False): + if str1 is None and str2 is None: + print("WARNING: Both None") + return True + if str1 is None or str2 is None: + return False + + str1, str2 = parse_math_answer(str1), parse_math_answer(str2) + + try: + ss1 = _strip_string(str1) + ss1 = postprocess(ss1) + ss2 = _strip_string(str2) + if verbose: + print(ss1, ss2) + return ss1 == ss2 + except Exception: + return str1 == str2 + + +def parse_math_answer(raw_string): + def remove_boxed(s): + left = "\\boxed{" + try: + assert s[: len(left)] == left + assert s[-1] == "}" + answer = s[len(left) : -1] + if "=" in answer: + answer = answer.split("=")[-1].lstrip(" ") + return answer + except Exception: + return None + + def last_boxed_only_string(string): + idx = string.rfind("\\boxed") + if idx < 0: + idx = string.rfind("\\fbox") + if idx < 0: + return None + i = idx + right_brace_idx = None + num_left_braces_open = 0 + while i < len(string): + if string[i] == "{": + num_left_braces_open += 1 + if string[i] == "}": + num_left_braces_open -= 1 + if num_left_braces_open == 0: + right_brace_idx = i + break + i += 1 + + if right_brace_idx is None: + retval = None + else: + retval = string[idx : right_brace_idx + 1] + + return retval + + def get_answer_with_dollar_sign(s): + first_pattern = "\$(.*)\$" + last_match = None + matches = re.findall(first_pattern, s) + if matches: + last_match = matches[-1] + if "=" in last_match: + last_match = last_match.split("=")[-1].lstrip(" ") + return last_match + + def get_answer_without_dollar_sign(s): + last_match = None + if "=" in s: + last_match = s.split("=")[-1].lstrip(" ").rstrip(".") + if "\\n" in last_match: + last_match = last_match.split("\\n")[0] + else: + pattern = "(?:\\$)?\d+(?:\.\d+)?(?![\w\d])" + matches = re.findall(pattern, s) + if matches: + last_match = matches[-1] + return last_match + + if "\\boxed" in raw_string: + answer = remove_boxed(last_boxed_only_string(raw_string)) + else: + answer = get_answer_with_dollar_sign(raw_string) + if not answer: + answer = get_answer_without_dollar_sign(raw_string) + return answer + + +# code from https://github.com/hendrycks/math/blob/main/modeling/math_equivalence.py +def _fix_fracs(string): + substrs = string.split("\\frac") + new_str = substrs[0] + if len(substrs) > 1: + substrs = substrs[1:] + for substr in substrs: + new_str += "\\frac" + if substr[0] == "{": + new_str += substr + else: + try: + assert len(substr) >= 2 + except Exception: + return string + a = substr[0] + b = substr[1] + if b != "{": + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}{" + b + "}" + post_substr + else: + new_str += "{" + a + "}{" + b + "}" + else: + if len(substr) > 2: + post_substr = substr[2:] + new_str += "{" + a + "}" + b + post_substr + else: + new_str += "{" + a + "}" + b + string = new_str + return string + + +def _fix_a_slash_b(string): + if len(string.split("/")) != 2: + return string + a = string.split("/")[0] + b = string.split("/")[1] + try: + a = int(a) + b = int(b) + assert string == "{}/{}".format(a, b) + new_string = "\\frac{" + str(a) + "}{" + str(b) + "}" + return new_string + except Exception: + return string + + +def _remove_right_units(string): + # "\\text{ " only ever occurs (at least in the val set) when describing units + if "\\text{ " in string: + splits = string.split("\\text{ ") + assert len(splits) == 2 + return splits[0] + else: + return string + + +def _fix_sqrt(string): + if "\\sqrt" not in string: + return string + splits = string.split("\\sqrt") + new_string = splits[0] + for split in splits[1:]: + if split[0] != "{": + a = split[0] + new_substr = "\\sqrt{" + a + "}" + split[1:] + else: + new_substr = "\\sqrt" + split + new_string += new_substr + return new_string + + +def _strip_string(string): + # linebreaks + string = string.replace("\n", "") + # print(string) + + # remove inverse spaces + string = string.replace("\\!", "") + # print(string) + + # replace \\ with \ + string = string.replace("\\\\", "\\") + # print(string) + + # replace tfrac and dfrac with frac + string = string.replace("tfrac", "frac") + string = string.replace("dfrac", "frac") + # print(string) + + # remove \left and \right + string = string.replace("\\left", "") + string = string.replace("\\right", "") + # print(string) + + # Remove circ (degrees) + string = string.replace("^{\\circ}", "") + string = string.replace("^\\circ", "") + + # remove dollar signs + string = string.replace("\\$", "") + + # remove units (on the right) + string = _remove_right_units(string) + + # remove percentage + string = string.replace("\\%", "") + string = string.replace("\%", "") + + # " 0." equivalent to " ." and "{0." equivalent to "{." Alternatively, add "0" if "." is the start of the string + string = string.replace(" .", " 0.") + string = string.replace("{.", "{0.") + # if empty, return empty string + if len(string) == 0: + return string + if string[0] == ".": + string = "0" + string + + # to consider: get rid of e.g. "k = " or "q = " at beginning + if len(string.split("=")) == 2: + if len(string.split("=")[0]) <= 2: + string = string.split("=")[1] + + # fix sqrt3 --> sqrt{3} + string = _fix_sqrt(string) + + # remove spaces + string = string.replace(" ", "") + + # \frac1b or \frac12 --> \frac{1}{b} and \frac{1}{2}, etc. Even works with \frac1{72} (but not \frac{72}1). Also does a/b --> \\frac{a}{b} + string = _fix_fracs(string) + + # manually change 0.5 --> \frac{1}{2} + if string == "0.5": + string = "\\frac{1}{2}" + + # NOTE: X/Y changed to \frac{X}{Y} in dataset, but in simple cases fix in case the model output is X/Y + string = _fix_a_slash_b(string) + + return string diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/_hrm8k_en_yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/_hrm8k_en_yaml new file mode 100644 index 0000000000000000000000000000000000000000..18c53d22997061ed43854d449e5b9e64a80e7335 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/_hrm8k_en_yaml @@ -0,0 +1,22 @@ +dataset_path: HAERAE-HUB/HRM8K +output_type: generate_until +test_split: test +doc_to_text: !function utils.doc_to_text +doc_to_target: !function utils.doc_to_target +process_results: !function utils.process_results +num_fewshot: 0 +generation_kwargs: + until: + - "" + - "<|end_of_text|>" + - "<|endoftext|>" + - "<|im_end|>" + max_gen_toks: 512 + do_sample: false + temperature: 0 +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_gsm8k_en.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_gsm8k_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2697a0b8284e3ec72267c46c5094531ccb389f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_gsm8k_en.yaml @@ -0,0 +1,3 @@ +include: _hrm8k_en_yaml +dataset_name: GSM8K +task: hrm8k_gsm8k_en diff --git a/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_mmmlu_en.yaml b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_mmmlu_en.yaml new file mode 100644 index 0000000000000000000000000000000000000000..812f62e268278cb52178aecd364b4329b7aec4c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/hrm8k/en/hrm8k_mmmlu_en.yaml @@ -0,0 +1,4 @@ +include: _hrm8k_en_yaml +dataset_name: MMMLU +task: hrm8k_mmmlu_en +doc_to_text: !function utils.doc_to_text_mmmlu diff --git a/lm-evaluation-harness/lm_eval/tasks/humaneval/humaneval.yaml b/lm-evaluation-harness/lm_eval/tasks/humaneval/humaneval.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e3a8d6d30ed9c312c9eac4e20deba4a6cc61510 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/humaneval/humaneval.yaml @@ -0,0 +1,30 @@ +task: humaneval +dataset_path: openai/openai_humaneval +unsafe_code: true +output_type: generate_until +test_split: test +doc_to_text: "{{prompt}}" +doc_to_target: "{{test}}\ncheck({{entry_point}})" +metric_list: + - metric: !function utils.pass_at_k + aggregation: mean + higher_is_better: true + k: [1] +generation_kwargs: + until: + - "\nclass" + - "\ndef" + - "\n#" + - "\nif" + - "\nprint" + max_gen_toks: 1024 + do_sample: false +repeats: 1 +num_fewshot: 0 +filter_list: + - name: "create_test" + filter: + - function: "custom" + filter_fn: !function utils.build_predictions +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/_include_base_44_croatian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/_include_base_44_croatian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..157850d78ec3a753f5c25aabd4b67bd8a25e4763 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/_include_base_44_croatian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_croatian +task: +- include_base_44_croatian_stem +- include_base_44_croatian_arts_humanities +- include_base_44_croatian_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e81897cb25ff7b7ffd910cec2b7cfe8e5cc3467 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_social_science.yaml @@ -0,0 +1,4 @@ +include: _croatian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_croatian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4a3693647e25a41b5b0184f5a07950e378736aa9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Croatian/include_base_44_croatian_stem.yaml @@ -0,0 +1,4 @@ +include: _croatian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_croatian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/_dutch_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/_dutch_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6aebdef762b40d9af6da20a14cf882c40c2790e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/_dutch_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Dutch +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0289fc60e2e01fa23b7b9333e3bab434be1cc842 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_applied_science.yaml @@ -0,0 +1,4 @@ +include: _dutch_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_dutch_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..deebc60dde9eb8ce89624b468f54fdd3c364feb8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _dutch_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_dutch_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c1635229dcd54d0e0ea8d98271f773614e11887 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/include_base_44_dutch_social_science.yaml @@ -0,0 +1,4 @@ +include: _dutch_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_dutch_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Dutch/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/_estonian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/_estonian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8d1a6246ce099a6bfe4940924f41d3614bfbb47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/_estonian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Estonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f88eedc1ba73ae7643cfc6dbb9305c6d5994e559 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_applied_science.yaml @@ -0,0 +1,4 @@ +include: _estonian_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_estonian_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ba29726c19ac61605ec4b1f467bae88b8b160754 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_social_science.yaml @@ -0,0 +1,4 @@ +include: _estonian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_estonian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dcf80871829f60b8dd9283259c170215b4c4574c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/include_base_44_estonian_stem.yaml @@ -0,0 +1,4 @@ +include: _estonian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_estonian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Estonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/_include_base_44_finnish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/_include_base_44_finnish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..831c8d9a73768479f7921f2fca921aff3e2aed73 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/_include_base_44_finnish.yaml @@ -0,0 +1,12 @@ +group: include_base_44_finnish +task: +- include_base_44_finnish_stem +- include_base_44_finnish_arts_humanities +- include_base_44_finnish_social_science +- include_base_44_finnish_health_oriented_education +- include_base_44_finnish_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50ec3d3416aa50b467134bc0db0660373acb16e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_applied_science.yaml @@ -0,0 +1,4 @@ +include: _finnish_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_finnish_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5e2815d68b75711937c7b57c7455335fb68035f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _finnish_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_finnish_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf8d9b7e91932163a5931410a557f8e6f2e81a86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _finnish_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_finnish_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41f447b6b97c46f812f0221dde214a2d173fbbb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_social_science.yaml @@ -0,0 +1,4 @@ +include: _finnish_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_finnish_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5056056245dd2d8d84b70248e18bf86a458c6ef3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/include_base_44_finnish_stem.yaml @@ -0,0 +1,4 @@ +include: _finnish_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_finnish_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Finnish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/_french_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/_french_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c11e99be2f82824aa12c56f665f8a66a707a4539 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/_french_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: French +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/_include_base_44_french.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/_include_base_44_french.yaml new file mode 100644 index 0000000000000000000000000000000000000000..40012728edbc6e4ce931331eb630adcde7ac6bda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/_include_base_44_french.yaml @@ -0,0 +1,12 @@ +group: include_base_44_french +task: +- include_base_44_french_stem +- include_base_44_french_social_science +- include_base_44_french_health_oriented_education +- include_base_44_french_arts_humanities +- include_base_44_french_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..62153e34984329b52a7d96253d57ab909463e221 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _french_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_french_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a8b9b12ac125631b24baddd8988db0ade36b5bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_driving_license.yaml @@ -0,0 +1,4 @@ +include: _french_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_french_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..be22467a701d509aa4a57635f7f1f3f41b1b11f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _french_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_french_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e1efa37dfdd58ab221f7a05f903adb9b787f9eae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_social_science.yaml @@ -0,0 +1,4 @@ +include: _french_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_french_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aee5db150e4c7fb7094f9be727c23e5991574eae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/include_base_44_french_stem.yaml @@ -0,0 +1,4 @@ +include: _french_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_french_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/French/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/French/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/French/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_georgian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_georgian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e827d49433a43f554715863cf8e3cd396be857ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_georgian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Georgian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_include_base_44_georgian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_include_base_44_georgian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16b1306ff6f727a48a4aa9c2ef1c175747ea54f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/_include_base_44_georgian.yaml @@ -0,0 +1,8 @@ +group: include_base_44_georgian +task: +- include_base_44_georgian_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/include_base_44_georgian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/include_base_44_georgian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0b049fc7bec5af18e8fb9ff52870cd60a901fa1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/include_base_44_georgian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _georgian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_georgian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Georgian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/_german_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/German/_german_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..914f2ab8c77424fb8b615467278296dad47dd6c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/_german_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: German +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/_include_base_44_german.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/German/_include_base_44_german.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c06efcb6422cb53f16b0e5df069e5b2085164909 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/_include_base_44_german.yaml @@ -0,0 +1,10 @@ +group: include_base_44_german +task: +- include_base_44_german_stem +- include_base_44_german_social_science +- include_base_44_german_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3ac238dd28f7a261c1f7451106849dc48d8f2b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_driving_license.yaml @@ -0,0 +1,4 @@ +include: _german_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_german_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8faa76339ae160c971f856c22a6f87720acfd7d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_social_science.yaml @@ -0,0 +1,4 @@ +include: _german_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_german_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..06ed2e7ba42c24d572d72a06fba1422650cc62d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/include_base_44_german_stem.yaml @@ -0,0 +1,4 @@ +include: _german_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_german_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/German/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/German/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/German/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_greek_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_greek_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7a822fc2f49ccd11d801864a434b5d23675e8b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_greek_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Greek +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_include_base_44_greek.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_include_base_44_greek.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77772f269eb3ad9f79e12a84e62c6fe201094f9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/_include_base_44_greek.yaml @@ -0,0 +1,14 @@ +group: include_base_44_greek +task: +- include_base_44_greek_stem +- include_base_44_greek_arts_humanities +- include_base_44_greek_social_science +- include_base_44_greek_business_commerce +- include_base_44_greek_health_oriented_education +- include_base_44_greek_professional_certification +- include_base_44_greek_medical_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0f58aa0d3e6957c1e19f8a751aedbf22cd4a9d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_greek_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f10daa10a0316454a3c890de8de0702c904d06dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_greek_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a0d0e15d9f0ec4e7ceb40ae6f515104e078df78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_greek_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7607dc9e7e874476077cbf9882963973e0185f3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_medical_license.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Medical License. +process_docs: !function 'utils.process_medical_license' +task: include_base_44_greek_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b62bf01fec370e8737c9d6a63209283810914bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_greek_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..efe82841059c3b944f43e31f5a762c88163d3d2c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_social_science.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_greek_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d5e56e27e0afa8ddfab9477baae493956ba644d6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/include_base_44_greek_stem.yaml @@ -0,0 +1,4 @@ +include: _greek_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_greek_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Greek/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_hebrew_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_hebrew_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc6fcfa2e4ad9555b382ab58009a847a42e2cf1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_hebrew_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hebrew +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_include_base_44_hebrew.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_include_base_44_hebrew.yaml new file mode 100644 index 0000000000000000000000000000000000000000..36f44731c22b6800d7a9174712f5470b8734626e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/_include_base_44_hebrew.yaml @@ -0,0 +1,9 @@ +group: include_base_44_hebrew +task: +- include_base_44_hebrew_arts_humanities +- include_base_44_hebrew_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1bbe72365b248c52ab933f1467bb2fb1a992655 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _hebrew_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hebrew_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3308e8ab4ad92a1f5e84720e479fcba4e938ee03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/include_base_44_hebrew_driving_license.yaml @@ -0,0 +1,4 @@ +include: _hebrew_template_yaml +description: The following is multiple-choice question about Driving license. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hebrew_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hebrew/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_hindi_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_hindi_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf6ca8c1e94d002ebb64fec867d30694ff2efc81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_hindi_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hindi +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_include_base_44_hindi.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_include_base_44_hindi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..603f14c982b0032bf0865204905baf95a08409a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/_include_base_44_hindi.yaml @@ -0,0 +1,15 @@ +group: include_base_44_hindi +task: +- include_base_44_hindi_professional_certification +- include_base_44_hindi_stem +- include_base_44_hindi_social_science +- include_base_44_hindi_driving_license +- include_base_44_hindi_applied_science +- include_base_44_hindi_arts_humanities +- include_base_44_hindi_general_knowledge +- include_base_44_hindi_health_oriented_education +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e79ca38b2ca350b186d9f68b370fd3049fc1a2ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_applied_science.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hindi_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19fb26c7d40cbd263800e06f402506eaecab02d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hindi_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5518ec4df43923b97366d7bb48afe68e0d46e088 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_driving_license.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hindi_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f1af9151b0d2d874369a6034b75987f67ae9d1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_general_knowledge.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about General knowledge. +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_hindi_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0faad17dd25d346f950a306c77e9f4354d7cd9a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_hindi_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0610a13394a535309020cc9a60cd2fea5cb4b079 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_hindi_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb237f1dec8eca933fe0bf3079ea8ab664133fc1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_social_science.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_hindi_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f66b0e8a1d6846bbf96d05a0715262f9a1e20547 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/include_base_44_hindi_stem.yaml @@ -0,0 +1,4 @@ +include: _hindi_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_hindi_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hindi/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_hungarian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_hungarian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..47b30358909e8a9b3e2097202db96573634f49d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_hungarian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hungarian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_include_base_44_hungarian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_include_base_44_hungarian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a759e8f3e9f924445c863f84d8d72d1dac43cf24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/_include_base_44_hungarian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_hungarian +task: +- include_base_44_hungarian_stem +- include_base_44_hungarian_applied_science +- include_base_44_hungarian_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec2fde5aee59218f7d9a16955d894446b636f1e5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_applied_science.yaml @@ -0,0 +1,4 @@ +include: _hungarian_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hungarian_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7af2c1501bb88ed45370cb827bea00d5ac07aae1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_social_science.yaml @@ -0,0 +1,4 @@ +include: _hungarian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_hungarian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9f818226114a9a6dd60db124351cb85b3cdcd03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/include_base_44_hungarian_stem.yaml @@ -0,0 +1,4 @@ +include: _hungarian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_hungarian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Hungarian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_include_base_44_indonesian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_include_base_44_indonesian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f854e0a769e67311043ae2ef964f60f0963c4581 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_include_base_44_indonesian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_indonesian +task: +- include_base_44_indonesian_arts_humanities +- include_base_44_indonesian_social_science +- include_base_44_indonesian_stem +- include_base_44_indonesian_applied_science +- include_base_44_indonesian_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_indonesian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_indonesian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..eed3a59c1b3ff5717e07aad4acd3ab6a4a1fce23 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/_indonesian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Indonesian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7be50328566fd7124f415f5830f5d0c51198c964 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_applied_science.yaml @@ -0,0 +1,4 @@ +include: _indonesian_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_indonesian_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e439865e7651203a52b33760cea87c672296df88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _indonesian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_indonesian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..960f4fd7eea40305f6bd612b6d628ebb07873bb0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _indonesian_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_indonesian_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0bf90151fca9b16bbd4eb4a7d9d940c22a32ed3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_social_science.yaml @@ -0,0 +1,4 @@ +include: _indonesian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_indonesian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6034292b721e033cd8941ff5d0c0cb18d95fbd9f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/include_base_44_indonesian_stem.yaml @@ -0,0 +1,4 @@ +include: _indonesian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_indonesian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Indonesian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_include_base_44_italian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_include_base_44_italian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20fb8a54f63457414b8286e1a3396ad0cc9e098a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_include_base_44_italian.yaml @@ -0,0 +1,13 @@ +group: include_base_44_italian +task: +- include_base_44_italian_stem +- include_base_44_italian_arts_humanities +- include_base_44_italian_social_science +- include_base_44_italian_applied_science +- include_base_44_italian_health_oriented_education +- include_base_44_italian_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_italian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_italian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8bd03586e0eb9a4e3cf4121af6e409df444ff88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/_italian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Italian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76cb047785817a0223bfec4563e54262c2350ac1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_applied_science.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_italian_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a88361d40b7d911c69d6abe319e055da77b87b45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_italian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..852d84c84747d47993ed977ab36a370e4ed203d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_italian_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09d789465ec014de08a8d963c375b88679357301 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_italian_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fd9b7ffd81cac603fa42b18ce708c489bb76e94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_social_science.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_italian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..548fcd52f2efa86f13e43ccaa185e2216ee21603 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/include_base_44_italian_stem.yaml @@ -0,0 +1,4 @@ +include: _italian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_italian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Italian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_include_base_44_japanese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_include_base_44_japanese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4dddddc2204fb281a6a765ebbc2211a859523688 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_include_base_44_japanese.yaml @@ -0,0 +1,10 @@ +group: include_base_44_japanese +task: +- include_base_44_japanese_driving_license +- include_base_44_japanese_medical_license +- include_base_44_japanese_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_japanese_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_japanese_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..878d1a96323249d329613c541aa3326c5e948c16 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/_japanese_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Japanese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9245f088586fcfd78b66855d70eb6f2bb003b82a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_driving_license.yaml @@ -0,0 +1,4 @@ +include: _japanese_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_japanese_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ba01c5b05def3e214f8ae82b5f01eaac67b89b4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_medical_license.yaml @@ -0,0 +1,4 @@ +include: _japanese_template_yaml +description: The following is multiple-choice question about Medical License. +process_docs: !function 'utils.process_medical_license' +task: include_base_44_japanese_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..640e00ed62147fdd293ff04b8934be5b3309d1dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/include_base_44_japanese_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _japanese_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_japanese_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Japanese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_include_base_44_kazakh.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_include_base_44_kazakh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7cf3c2e89280260cfca1b4493320c3a4e528a469 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_include_base_44_kazakh.yaml @@ -0,0 +1,8 @@ +group: include_base_44_kazakh +task: +- include_base_44_kazakh_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_kazakh_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_kazakh_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a800eeb37b1675b576a69cbc3267c9e3c9cc86bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/_kazakh_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Kazakh +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/include_base_44_kazakh_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/include_base_44_kazakh_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8427d8a4a54c57a6a76633bbfdb2c8f6467dad2a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/include_base_44_kazakh_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _kazakh_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_kazakh_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Kazakh/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_include_base_44_korean.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_include_base_44_korean.yaml new file mode 100644 index 0000000000000000000000000000000000000000..41e491db3d0f0fd217b88e2c6ef96d4d61f01eac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_include_base_44_korean.yaml @@ -0,0 +1,9 @@ +group: include_base_44_korean +task: +- include_base_44_korean_professional_certification +- include_base_44_korean_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_korean_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_korean_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd5d2e07e2baea8f60c0b3ab9d9c80aeaf2ba8d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/_korean_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Korean +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..94d360c408eed0435e10eddee0f1ae5da0f0a028 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _korean_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_korean_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7ab6a7e6f2f395021b9e4da65dddc9373c85891e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/include_base_44_korean_social_science.yaml @@ -0,0 +1,4 @@ +include: _korean_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_korean_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Korean/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_include_base_44_lithuanian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_include_base_44_lithuanian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..897d40bdb5ccf25efa3cec095f27164af3699f26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_include_base_44_lithuanian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_lithuanian +task: +- include_base_44_lithuanian_arts_humanities +- include_base_44_lithuanian_stem +- include_base_44_lithuanian_social_science +- include_base_44_lithuanian_business_commerce +- include_base_44_lithuanian_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_lithuanian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_lithuanian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..64315ee76457e6fa7e94c34e9d27be8808c3e1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/_lithuanian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Lithuanian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff33d2481638bff772e90f7a3c924663730608db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_lithuanian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6d43b377f2da2207af97c2a478d2af8d53d8e052 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_lithuanian_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..643b46027f79ae9a8d7a3f6d9fe939340c5735eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_lithuanian_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f62b27fed7876081dfadcd88dbb96ff2784a3ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_social_science.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_lithuanian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b408a83576736cbf53d7ed511c48044dc2e9e07d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/include_base_44_lithuanian_stem.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_lithuanian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Lithuanian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_include_base_44_malay.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_include_base_44_malay.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ad225b54d21dc97b185d2f83b55327d96b214282 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_include_base_44_malay.yaml @@ -0,0 +1,10 @@ +group: include_base_44_malay +task: +- include_base_44_malay_social_science +- include_base_44_malay_business_commerce +- include_base_44_malay_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_malay_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_malay_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd1a176785b9cdb5d0dcee947010f71d442a7423 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/_malay_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malay +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..caf255d41e606ca582ac46618c1f690e2b02bc81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _malay_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malay_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a9d39eadaca8e57361cb69dc95f33677d1257d80 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _malay_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_malay_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..117947e10d63d758c6bd351cb53bb3004f977643 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/include_base_44_malay_social_science.yaml @@ -0,0 +1,4 @@ +include: _malay_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_malay_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malay/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_include_base_44_malayalam.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_include_base_44_malayalam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b4ba552e2d2cea21751065449e7fd6961c068995 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_include_base_44_malayalam.yaml @@ -0,0 +1,13 @@ +group: include_base_44_malayalam +task: +- include_base_44_malayalam_stem +- include_base_44_malayalam_marine_license +- include_base_44_malayalam_health_oriented_education +- include_base_44_malayalam_arts_humanities +- include_base_44_malayalam_social_science +- include_base_44_malayalam_general_knowledge +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_malayalam_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_malayalam_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..62355b0098c9dbc9aca70e081214193483c1e044 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/_malayalam_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malayalam +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df20c8e14766a71cc08582e67be9a017cb7102e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malayalam_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d42ea34c70916343f1ccd3f311bac52965a4249 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_general_knowledge.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about General knowledge. +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_malayalam_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0060ecf2b6c9bc8b36d9eca18618bbc3ed4c9005 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_malayalam_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_marine_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_marine_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8cb75de8cf438122948cba213219778cb0b84fb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_marine_license.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about Marine License. +process_docs: !function 'utils.process_marine_license' +task: include_base_44_malayalam_marine_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b4faf9f4591b92c6c6029ab7e46edf27cd314db --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_social_science.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_malayalam_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90b66cdb5d58c3e22236b4e8df37b54e5df58c9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/include_base_44_malayalam_stem.yaml @@ -0,0 +1,4 @@ +include: _malayalam_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_malayalam_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Malayalam/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_include_base_44_nepali.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_include_base_44_nepali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..94f6c278c72456cacc25b1ed95ad7c2983779c1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_include_base_44_nepali.yaml @@ -0,0 +1,9 @@ +group: include_base_44_nepali +task: +- include_base_44_nepali_driving_license +- include_base_44_nepali_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_nepali_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_nepali_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3dd04705a7c11f0a07c683f6a5a2a0149ba6961 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/_nepali_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Nepali +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c9da29a7c6579d8ef4ac321f1c61246af0c73826 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_driving_license.yaml @@ -0,0 +1,4 @@ +include: _nepali_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_nepali_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..026217ac0a7bf1b86f05a3933681682b7587f747 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/include_base_44_nepali_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _nepali_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_nepali_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Nepali/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_include_base_44_north macedonian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_include_base_44_north macedonian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..776e3310e55ff88211c53ed2d1a2748e6fcb2f73 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_include_base_44_north macedonian.yaml @@ -0,0 +1,11 @@ +group: include_base_44_north macedonian +task: +- include_base_44_north macedonian_arts_humanities +- include_base_44_north macedonian_stem +- include_base_44_north macedonian_business_commerce +- include_base_44_north macedonian_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_north macedonian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_north macedonian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..70cf2a749efe48b164ac1be21af65bdef8fbe508 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/_north macedonian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: North Macedonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..079d3cc2a4e5bf4dc4674e1e603a9332d1c7037c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_north macedonian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c6529103274489eab77b7e14fa3ba1e89ac2386 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_north macedonian_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3c63ac3eda6f70c1dea4b03480ebd200d560a21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_social_science.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_north macedonian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4056924ad27d7d17c59f0051b1269a9e16156e03 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/include_base_44_north macedonian_stem.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_north macedonian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/North Macedonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_include_base_44_persian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_include_base_44_persian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5be0d3cb4f452c08b4346e774aebdf397ab6642d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_include_base_44_persian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_persian +task: +- include_base_44_persian_arts_humanities +- include_base_44_persian_social_science +- include_base_44_persian_professional_certification +- include_base_44_persian_stem +- include_base_44_persian_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_persian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_persian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4e4eec0811a1ef02b5002f39fa0f2cef387a975 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/_persian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Persian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2848e8f837205e53faf04c7fab923155a1880eea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _persian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_persian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cafbba46c711ff5143de63a3e6a1696182e2eac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_driving_license.yaml @@ -0,0 +1,4 @@ +include: _persian_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_persian_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6dc6f1805592ce18908486a5a89b7d634fbfd794 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _persian_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_persian_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f6333c4f2c0fc120efc0c3ff683c141adc6788f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_social_science.yaml @@ -0,0 +1,4 @@ +include: _persian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_persian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b7ede848a080eca09a574972757cfff8de54b4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/include_base_44_persian_stem.yaml @@ -0,0 +1,4 @@ +include: _persian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_persian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Persian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_include_base_44_polish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_include_base_44_polish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cdd4ce191ad39920b626667b5fb8db48a81f51f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_include_base_44_polish.yaml @@ -0,0 +1,10 @@ +group: include_base_44_polish +task: +- include_base_44_polish_professional_certification +- include_base_44_polish_social_science +- include_base_44_polish_stem +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_polish_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_polish_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..144716e8d8caa4ff6243c4b81a9daaadc4aa33fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/_polish_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Polish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb8505e2dadd036ed03409e2f2035e49eb4bbb68 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _polish_template_yaml +description: The following is multiple-choice question about Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_polish_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22724778d8b8b1f06f6389af9df80785b8c1e1a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_social_science.yaml @@ -0,0 +1,4 @@ +include: _polish_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_polish_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38fe2cb007fab17bd720cd17a808450f69275a88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/include_base_44_polish_stem.yaml @@ -0,0 +1,4 @@ +include: _polish_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_polish_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Polish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_include_base_44_portuguese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_include_base_44_portuguese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1140efb9d97c985f22396d89da89a72c4b9f8f2b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_include_base_44_portuguese.yaml @@ -0,0 +1,13 @@ +group: include_base_44_portuguese +task: +- include_base_44_portuguese_stem +- include_base_44_portuguese_social_science +- include_base_44_portuguese_arts_humanities +- include_base_44_portuguese_health_oriented_education +- include_base_44_portuguese_business_commerce +- include_base_44_portuguese_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_portuguese_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_portuguese_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ae882b22cd6f4a48c89737dfb7acc19b53add17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/_portuguese_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Portuguese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9ea01e614dff6a7753e123ca401aea4d69e0f098 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_applied_science.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_portuguese_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..011cd16fe850fc838151d29d976d7c9df0130ae2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_portuguese_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8e7f8f67ca83808c64edbdff7395ceebebc3c30 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_portuguese_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a81491e90dd64e5a34ed610eda7675397a0b730 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_portuguese_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd240be79a7fb31ce4b952ca0b313f0df993705b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_social_science.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_portuguese_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7db9dd9134e54f94dac5d6fe9c4cac526580c64a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/include_base_44_portuguese_stem.yaml @@ -0,0 +1,4 @@ +include: _portuguese_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_portuguese_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Portuguese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_include_base_44_russian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_include_base_44_russian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4b36d155db795cf7ab426489e341860ca7c0d6e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_include_base_44_russian.yaml @@ -0,0 +1,15 @@ +group: include_base_44_russian +task: +- include_base_44_russian_stem +- include_base_44_russian_health_oriented_education +- include_base_44_russian_arts_humanities +- include_base_44_russian_driving_license +- include_base_44_russian_social_science +- include_base_44_russian_business_commerce +- include_base_44_russian_marine_license +- include_base_44_russian_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_russian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_russian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b0e24871dfd767ebaa38747e9942f36c6222461 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/_russian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Russian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a94f59984870089d8b3b92fd3a4a8c56806439c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_applied_science.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_russian_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2301a96d6bda5d0530ba0958299532fb1df8afbe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_russian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e848acc4f4da1463ff45fb9240cb8f6df7209378 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_russian_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e67e062046804ea4f34af7ee6c525f3756af40ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_driving_license.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_russian_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..df168d7eb7894a38b7b1db266120a0a8b31c85ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_russian_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_marine_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_marine_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa9bde44db5408ea07582a4d0824a2421c5fa746 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_marine_license.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Marine License. +process_docs: !function 'utils.process_marine_license' +task: include_base_44_russian_marine_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e51af6f33d33d69d5e174f60710ae6eb3f0b68f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_social_science.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_russian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e03df575ccd835e8bcb4f5a16cd872f4b9603511 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/include_base_44_russian_stem.yaml @@ -0,0 +1,4 @@ +include: _russian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_russian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Russian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_include_base_44_serbian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_include_base_44_serbian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d183634572e578d2a1dcb0891ceacb5e71068ffd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_include_base_44_serbian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_serbian +task: +- include_base_44_serbian_stem +- include_base_44_serbian_arts_humanities +- include_base_44_serbian_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_serbian_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_serbian_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b413a08ad6dd70cf46a1272743175fc1d0153fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/_serbian_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Serbian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..46ae9cc05e336ef96121b978e2386ad04796da18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _serbian_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_serbian_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8aea79814e1293c6bfba700861dd139110f710a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_social_science.yaml @@ -0,0 +1,4 @@ +include: _serbian_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_serbian_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a97964c10cf061bd6e5314326f74697408f696cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/include_base_44_serbian_stem.yaml @@ -0,0 +1,4 @@ +include: _serbian_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_serbian_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Serbian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_include_base_44_spanish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_include_base_44_spanish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d1b60c5eb11baf5aa820d27d3f3eafdee9d6dc6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_include_base_44_spanish.yaml @@ -0,0 +1,11 @@ +group: include_base_44_spanish +task: +- include_base_44_spanish_stem +- include_base_44_spanish_social_science +- include_base_44_spanish_arts_humanities +- include_base_44_spanish_health_oriented_education +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_spanish_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_spanish_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..40b8981d6488dd5c326d16096713b769c2214dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/_spanish_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Spanish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2e84414d7c374d77933a97351b95ca0cadee306 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _spanish_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_spanish_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50c24b6b19e593577b5262f8bae5ca439e57ffc5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _spanish_template_yaml +description: The following is multiple-choice question about Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_spanish_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5221a6b06bea95efad7f5a90e26625a4f6e00de2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_social_science.yaml @@ -0,0 +1,4 @@ +include: _spanish_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_spanish_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb2f12cfdee6ae7ba25d2cc3f8eeaac7655a5eb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/include_base_44_spanish_stem.yaml @@ -0,0 +1,4 @@ +include: _spanish_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_spanish_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Spanish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/_tagalog_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/_tagalog_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..88040e14d6f83717a48e22ed67c728bd2951f3c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/_tagalog_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Tagalog +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..056195732d1c1f0b27d11968cc6878d7ab4fee24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _tagalog_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_tagalog_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f7fc60deb05aafc05d3b75bcdbbdf340f7ad5431 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/include_base_44_tagalog_driving_license.yaml @@ -0,0 +1,4 @@ +include: _tagalog_template_yaml +description: The following is multiple-choice question about Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_tagalog_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tagalog/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/_include_base_44_tamil.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/_include_base_44_tamil.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f7527bbc7b0919ff59f7d16a780a598e49f1d1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/_include_base_44_tamil.yaml @@ -0,0 +1,9 @@ +group: include_base_44_tamil +task: +- include_base_44_tamil_stem +- include_base_44_tamil_general_knowledge +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48d7d7e00540f483c2435b96f7f04a0f890eeb21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_general_knowledge.yaml @@ -0,0 +1,4 @@ +include: _tamil_template_yaml +description: The following is multiple-choice question about General knowledge. +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_tamil_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e21eeeccb9f8d333c62bb3ee8f0860ca0586e34e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/include_base_44_tamil_stem.yaml @@ -0,0 +1,4 @@ +include: _tamil_template_yaml +description: The following is multiple-choice question about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_tamil_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Tamil/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/_include_base_44_telugu.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/_include_base_44_telugu.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32dd6b6ad8255dec373091aa04579c71daf55fbb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/_include_base_44_telugu.yaml @@ -0,0 +1,11 @@ +group: include_base_44_telugu +task: +- include_base_44_telugu_arts_humanities +- include_base_44_telugu_applied_science +- include_base_44_telugu_stem +- include_base_44_telugu_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f50ea1ccbf1b2fbaf1ebd48ff310e9ed433a995 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_applied_science.yaml @@ -0,0 +1,4 @@ +include: _telugu_template_yaml +description: The following is multiple-choice question about Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_telugu_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e0fb4925eb366769a6500dd1c6c316edb93c138d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Telugu/include_base_44_telugu_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _telugu_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_telugu_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_include_base_44_turkish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_include_base_44_turkish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15289b999198d7e30f13d00356bec87a4dd4f0df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_include_base_44_turkish.yaml @@ -0,0 +1,11 @@ +group: include_base_44_turkish +task: +- include_base_44_turkish_business_commerce +- include_base_44_turkish_stem +- include_base_44_turkish_arts_humanities +- include_base_44_turkish_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_turkish_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_turkish_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f42709007eddc3b172eea6f017ff24a0181b78c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/_turkish_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Turkish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a42decac5c078077012b5100319223d934193524 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _turkish_template_yaml +description: The following is multiple-choice question about Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_turkish_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0cd6f8e484cee2a814949056584d502470a725e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Turkish/include_base_44_turkish_social_science.yaml @@ -0,0 +1,4 @@ +include: _turkish_template_yaml +description: The following is multiple-choice question about Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_turkish_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/default/Ukrainian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/default/Ukrainian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/default/Ukrainian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9b3d1854e859fc750c08af1a3f90bbaef0eb501a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_applied_science.yaml @@ -0,0 +1,5 @@ +include: _chinese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_chinese_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ae4775ff4c612f9541432ce34e384de141186dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_driving_license.yaml @@ -0,0 +1,5 @@ +include: _chinese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_chinese_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29a0937bc81aa96faec569482a71155891858dec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_social_science.yaml @@ -0,0 +1,5 @@ +include: _chinese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_chinese_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d97e71e17eddff151013caccd047931dd42f7ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Chinese/include_base_44_chinese_stem.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_chinese_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/_include_base_44_croatian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/_include_base_44_croatian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a929cbccfde6ccd54d360935151e14c68fa7f901 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/_include_base_44_croatian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_croatian +task: +- include_base_44_croatian_few_shot_en_stem +- include_base_44_croatian_few_shot_en_arts_humanities +- include_base_44_croatian_few_shot_en_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..004557e5ff56ec5b832f6e4c000f6903bce74412 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_social_science.yaml @@ -0,0 +1,5 @@ +include: _croatian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_croatian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1aabf5e24ae68b7cbb29e1e524ba2685ad38281 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/include_base_44_croatian_stem.yaml @@ -0,0 +1,4 @@ +include: _croatian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_croatian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Croatian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_dutch_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_dutch_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6aebdef762b40d9af6da20a14cf882c40c2790e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_dutch_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Dutch +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_include_base_44_dutch.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_include_base_44_dutch.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aaa658141933e925a9edcfed1603f92e9318c43c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/_include_base_44_dutch.yaml @@ -0,0 +1,12 @@ +group: include_base_44_dutch +task: +- include_base_44_dutch_few_shot_en_arts_humanities +- include_base_44_dutch_few_shot_en_stem +- include_base_44_dutch_few_shot_en_social_science +- include_base_44_dutch_few_shot_en_health_oriented_education +- include_base_44_dutch_few_shot_en_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..139af0959e91e195a5941d5aa70e79617a940e95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_applied_science.yaml @@ -0,0 +1,5 @@ +include: _dutch_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_dutch_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e726521d8e37bd6f9f18e2bdaba785126c14220 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _dutch_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_dutch_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1d9ea2b89b53c1eac38dc0d774fd97e0bc6ad0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_social_science.yaml @@ -0,0 +1,5 @@ +include: _dutch_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_dutch_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a169506225249617280243c2aa7c0d75647100d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/include_base_44_dutch_stem.yaml @@ -0,0 +1,4 @@ +include: _dutch_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_dutch_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Dutch/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_estonian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_estonian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8d1a6246ce099a6bfe4940924f41d3614bfbb47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_estonian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Estonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_include_base_44_estonian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_include_base_44_estonian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0e48a183d3fa2b1b130463b488c034aaff67a424 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/_include_base_44_estonian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_estonian +task: +- include_base_44_estonian_few_shot_en_health_oriented_education +- include_base_44_estonian_few_shot_en_social_science +- include_base_44_estonian_few_shot_en_applied_science +- include_base_44_estonian_few_shot_en_stem +- include_base_44_estonian_few_shot_en_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c41317a35e9691369e3faa924bcaf3d183e613a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_estonian_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d00c0ca1852fad084244ff13fe36fb5945fd8f9a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_estonian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..25ad75a0dcb5741e1fe33a9fd599469e2f476605 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_estonian_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab29c22671af09c146117fb17a8ab481303d5a0e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_social_science.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_estonian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..72f5ac2d626c05be95b5153997101c9dfcdf7e1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/include_base_44_estonian_stem.yaml @@ -0,0 +1,4 @@ +include: _estonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_estonian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Estonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_finnish_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_finnish_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce7cc1f645feee31a8cf0472e65894127df83315 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_finnish_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Finnish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_include_base_44_finnish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_include_base_44_finnish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50df9484d91041b8ec2b4de311e2359455a78f74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/_include_base_44_finnish.yaml @@ -0,0 +1,12 @@ +group: include_base_44_finnish +task: +- include_base_44_finnish_few_shot_en_stem +- include_base_44_finnish_few_shot_en_arts_humanities +- include_base_44_finnish_few_shot_en_social_science +- include_base_44_finnish_few_shot_en_health_oriented_education +- include_base_44_finnish_few_shot_en_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24ebd9064e5473bc5a205abe0757077ddccaf1ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_applied_science.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_finnish_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2f04c3ba30eeb714ead23542dd0bc1a941ba3bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_finnish_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c14fe14f89ecf75c7d6cf67945ee1ff9ffbd1638 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_social_science.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_finnish_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c42e9ead85dbf13a210880045f8efb55b6f0c5b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/include_base_44_finnish_stem.yaml @@ -0,0 +1,4 @@ +include: _finnish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_finnish_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Finnish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_french_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_french_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c11e99be2f82824aa12c56f665f8a66a707a4539 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_french_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: French +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_include_base_44_french.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_include_base_44_french.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12ef5dc08149449cd5f3efc9feb0ff2374d8bb2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/_include_base_44_french.yaml @@ -0,0 +1,12 @@ +group: include_base_44_french +task: +- include_base_44_french_few_shot_en_stem +- include_base_44_french_few_shot_en_social_science +- include_base_44_french_few_shot_en_health_oriented_education +- include_base_44_french_few_shot_en_arts_humanities +- include_base_44_french_few_shot_en_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9db14bb2861016f896ae65e03b986e55d50afe07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_french_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d09730b9c6bba6cedb8ce4eaca4fe276b5817f12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_driving_license.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_french_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09ec7bc0aee11269060010b121786af03bcb0afd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_french_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1610f4b7f067796371ac503caa5543fc88f99cfa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_social_science.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_french_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c21d938f30e8c78adf9b310b54104619d40cf817 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/include_base_44_french_stem.yaml @@ -0,0 +1,4 @@ +include: _french_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_french_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/French/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_georgian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_georgian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e827d49433a43f554715863cf8e3cd396be857ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_georgian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Georgian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_include_base_44_georgian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_include_base_44_georgian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68174d23d40fb97cf2525f7fe9bc1efc0703a07d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/_include_base_44_georgian.yaml @@ -0,0 +1,8 @@ +group: include_base_44_georgian +task: +- include_base_44_georgian_few_shot_en_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/include_base_44_georgian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/include_base_44_georgian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..18ef9386eb5e2ab7956eba2da6043ff74bda63b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/include_base_44_georgian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _georgian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_georgian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Georgian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_german_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_german_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..914f2ab8c77424fb8b615467278296dad47dd6c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_german_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: German +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_include_base_44_german.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_include_base_44_german.yaml new file mode 100644 index 0000000000000000000000000000000000000000..77db3bf964a9b9f8ee10805b68f35f3c24adbcd3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/_include_base_44_german.yaml @@ -0,0 +1,10 @@ +group: include_base_44_german +task: +- include_base_44_german_few_shot_en_stem +- include_base_44_german_few_shot_en_social_science +- include_base_44_german_few_shot_en_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d24a1c28a7194be9f40a1211e9f9f3f60ba64fab --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_driving_license.yaml @@ -0,0 +1,5 @@ +include: _german_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_german_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..203f4e0dac8259d66c914676cae7205353949876 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_social_science.yaml @@ -0,0 +1,5 @@ +include: _german_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_german_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..539c14064e508f33ca69fc9c4e758a7e665a03f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/include_base_44_german_stem.yaml @@ -0,0 +1,4 @@ +include: _german_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_german_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/German/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_greek_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_greek_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7a822fc2f49ccd11d801864a434b5d23675e8b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_greek_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Greek +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_include_base_44_greek.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_include_base_44_greek.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8f256ee7ccd56e86baf1ab1f6841e4cbf950d78 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/_include_base_44_greek.yaml @@ -0,0 +1,14 @@ +group: include_base_44_greek +task: +- include_base_44_greek_few_shot_en_stem +- include_base_44_greek_few_shot_en_arts_humanities +- include_base_44_greek_few_shot_en_social_science +- include_base_44_greek_few_shot_en_business_commerce +- include_base_44_greek_few_shot_en_health_oriented_education +- include_base_44_greek_few_shot_en_professional_certification +- include_base_44_greek_few_shot_en_medical_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f7776418e051fc2f06e39b3fe80dc82b79922d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_greek_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..33cdbc5ef69baffffb66d62ccafdf4d60e08958b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_greek_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a6fb1ab997addfbe687e4a60ddcf9baf91371390 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_greek_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c05f93f7b25bc2f5c3366665a09a3ff70984f120 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_medical_license.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Medical + License. +process_docs: !function 'utils.process_medical_license' +task: include_base_44_greek_few_shot_en_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..260e85a68a95465e8e8a307078bbfbd9a36de7cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_greek_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6136469e503820838fb0f71b05ca8a3c54deed8d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_social_science.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_greek_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e2d41df74e6686442c9faf2edd370cd2e31f657 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/include_base_44_greek_stem.yaml @@ -0,0 +1,4 @@ +include: _greek_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_greek_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Greek/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_hebrew_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_hebrew_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc6fcfa2e4ad9555b382ab58009a847a42e2cf1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_hebrew_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hebrew +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_include_base_44_hebrew.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_include_base_44_hebrew.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ea16004e74f4289d09f41bb710b9e6653d1aeae7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/_include_base_44_hebrew.yaml @@ -0,0 +1,9 @@ +group: include_base_44_hebrew +task: +- include_base_44_hebrew_few_shot_en_arts_humanities +- include_base_44_hebrew_few_shot_en_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc70d8f5f43a34b280b4c65a7b3aa61cfbfab0c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _hebrew_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hebrew_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ef2d01ab00d23e2ed49a49045e3be7a79f8ab02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/include_base_44_hebrew_driving_license.yaml @@ -0,0 +1,5 @@ +include: _hebrew_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + license. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hebrew_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hebrew/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_hindi_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_hindi_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf6ca8c1e94d002ebb64fec867d30694ff2efc81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_hindi_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hindi +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_include_base_44_hindi.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_include_base_44_hindi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43037fc0e8f0f45c907723791e61b7858e84bef3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/_include_base_44_hindi.yaml @@ -0,0 +1,15 @@ +group: include_base_44_hindi +task: +- include_base_44_hindi_few_shot_en_professional_certification +- include_base_44_hindi_few_shot_en_stem +- include_base_44_hindi_few_shot_en_social_science +- include_base_44_hindi_few_shot_en_driving_license +- include_base_44_hindi_few_shot_en_applied_science +- include_base_44_hindi_few_shot_en_arts_humanities +- include_base_44_hindi_few_shot_en_general_knowledge +- include_base_44_hindi_few_shot_en_health_oriented_education +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0491c98ed7d7ff28a56bd06d656ef34a55582538 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_applied_science.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hindi_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..078c8ae22683fdc43a3d95a44033da45e7508578 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hindi_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d9c91799670b56aaa90d390ab0d19c4354c119a0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_driving_license.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hindi_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e03356999b9cc070c89a5a65a167c0db89ad1d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_general_knowledge.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about General + knowledge. +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_hindi_few_shot_en_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ece8a64fdf927bbdf196f9d6967f6489d09f5155 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_hindi_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a63361d2cce5adafaa9d8db8f0210c9ad8f9afdc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_hindi_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a0296e2a61c7a84cad191e6e00f65fea540edae6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_social_science.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_hindi_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e5baccb2a403c655db02a87c76412d52b631b31f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/include_base_44_hindi_stem.yaml @@ -0,0 +1,4 @@ +include: _hindi_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_hindi_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hindi/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_hungarian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_hungarian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..47b30358909e8a9b3e2097202db96573634f49d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_hungarian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hungarian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_include_base_44_hungarian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_include_base_44_hungarian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6aa05f56438c4ce35b871ca46d67c59ad7330905 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/_include_base_44_hungarian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_hungarian +task: +- include_base_44_hungarian_few_shot_en_stem +- include_base_44_hungarian_few_shot_en_applied_science +- include_base_44_hungarian_few_shot_en_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5157d552115a89f00736272ec42959c15c8227ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _hungarian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hungarian_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c73fcc13d696a94d32b8b52e04248c8454407e60 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_social_science.yaml @@ -0,0 +1,5 @@ +include: _hungarian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_hungarian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f5d95ec08fba35ad53a2918c1cd01509874e483 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/include_base_44_hungarian_stem.yaml @@ -0,0 +1,4 @@ +include: _hungarian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_hungarian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Hungarian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_include_base_44_indonesian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_include_base_44_indonesian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c4a2df61fa45f7af9db0303d7f3c2c2344e2bb2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_include_base_44_indonesian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_indonesian +task: +- include_base_44_indonesian_few_shot_en_arts_humanities +- include_base_44_indonesian_few_shot_en_social_science +- include_base_44_indonesian_few_shot_en_stem +- include_base_44_indonesian_few_shot_en_applied_science +- include_base_44_indonesian_few_shot_en_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_indonesian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_indonesian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..eed3a59c1b3ff5717e07aad4acd3ab6a4a1fce23 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/_indonesian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Indonesian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00f94ea6548d44810a7c738507607fc9f45a0660 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_indonesian_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ced0ce39fa012651af117419fdda2359d9caea22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_indonesian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5070917adb6e9440c3a6a5794e5d5f05f693015f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_indonesian_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f9aef592a505316f706c6ae48199ebe5b317901 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_social_science.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_indonesian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2006d13517cc6431f57634fc4c2d55861f23d43b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/include_base_44_indonesian_stem.yaml @@ -0,0 +1,4 @@ +include: _indonesian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_indonesian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Indonesian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_include_base_44_italian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_include_base_44_italian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9934f827e53a2d5d09365f5d5c9fff7655e3e846 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_include_base_44_italian.yaml @@ -0,0 +1,13 @@ +group: include_base_44_italian +task: +- include_base_44_italian_few_shot_en_stem +- include_base_44_italian_few_shot_en_arts_humanities +- include_base_44_italian_few_shot_en_social_science +- include_base_44_italian_few_shot_en_applied_science +- include_base_44_italian_few_shot_en_health_oriented_education +- include_base_44_italian_few_shot_en_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_italian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_italian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8bd03586e0eb9a4e3cf4121af6e409df444ff88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/_italian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Italian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa696cae483c5ac921b7ef0b18431ede469f5ece --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_italian_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7aeb966110c90c7f9a2932972872c646d40e3e2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_italian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e8f6825316eb35e25d5cbc1238e57755c5e882d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_italian_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3314177e50096dd43a44c8f7007f3703b0a75145 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_italian_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb430f3bc55dbca918ee7a825cf6fe4bcad80b6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_social_science.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_italian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6c14ceb9c964861dce8b18246078eb43763aedb9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/include_base_44_italian_stem.yaml @@ -0,0 +1,4 @@ +include: _italian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_italian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Italian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_include_base_44_japanese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_include_base_44_japanese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6accc3f43ee9e3fbaf41f694993e7cc8e3d66b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_include_base_44_japanese.yaml @@ -0,0 +1,10 @@ +group: include_base_44_japanese +task: +- include_base_44_japanese_few_shot_en_driving_license +- include_base_44_japanese_few_shot_en_medical_license +- include_base_44_japanese_few_shot_en_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_japanese_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_japanese_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..878d1a96323249d329613c541aa3326c5e948c16 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/_japanese_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Japanese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7aaefe79c3a88ba15d6672b489953b8f9001a9d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_driving_license.yaml @@ -0,0 +1,5 @@ +include: _japanese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_japanese_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f56f72fb005d46583a07a2e6f8e8551b2cb0c0c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_medical_license.yaml @@ -0,0 +1,5 @@ +include: _japanese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Medical + License. +process_docs: !function 'utils.process_medical_license' +task: include_base_44_japanese_few_shot_en_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1db58ab8ae1fed4c6f584070a64131e0e033b5b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/include_base_44_japanese_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _japanese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_japanese_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Japanese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_include_base_44_kazakh.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_include_base_44_kazakh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..37589e497dc61e8ee8e74cc7ff691db308a8763e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_include_base_44_kazakh.yaml @@ -0,0 +1,8 @@ +group: include_base_44_kazakh +task: +- include_base_44_kazakh_few_shot_en_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_kazakh_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_kazakh_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a800eeb37b1675b576a69cbc3267c9e3c9cc86bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/_kazakh_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Kazakh +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/include_base_44_kazakh_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/include_base_44_kazakh_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..94278d5444c171c240cc926ca84299f05dc72541 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/include_base_44_kazakh_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _kazakh_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_kazakh_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Kazakh/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_include_base_44_korean.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_include_base_44_korean.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5dc56865ad0651ecafde994a1d28d9db13db502 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_include_base_44_korean.yaml @@ -0,0 +1,9 @@ +group: include_base_44_korean +task: +- include_base_44_korean_few_shot_en_professional_certification +- include_base_44_korean_few_shot_en_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_korean_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_korean_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd5d2e07e2baea8f60c0b3ab9d9c80aeaf2ba8d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/_korean_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Korean +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b693df9b4e43cd7cb3149ec66f194653ce0fef0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _korean_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_korean_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1015c3572a350f848fd4366995155d5a5e37deef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/include_base_44_korean_social_science.yaml @@ -0,0 +1,5 @@ +include: _korean_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_korean_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Korean/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_include_base_44_lithuanian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_include_base_44_lithuanian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5d105f7c7015156f198830ce850457628ed1c778 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_include_base_44_lithuanian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_lithuanian +task: +- include_base_44_lithuanian_few_shot_en_arts_humanities +- include_base_44_lithuanian_few_shot_en_stem +- include_base_44_lithuanian_few_shot_en_social_science +- include_base_44_lithuanian_few_shot_en_business_commerce +- include_base_44_lithuanian_few_shot_en_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_lithuanian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_lithuanian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..64315ee76457e6fa7e94c34e9d27be8808c3e1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/_lithuanian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Lithuanian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8dd8f741fae260400bb580de5588ecde43f50d90 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_lithuanian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b5a890dc663a03bcac9b4edca94943c0ad7a064 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_lithuanian_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e3c9219f272134d4843760156a8f3794a86d18f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_lithuanian_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a561858edd428fae01caca50c4ddccf0e26b478e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_social_science.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_lithuanian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fda5e9cc856fa77caa0911d4e9e8927ae144800b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/include_base_44_lithuanian_stem.yaml @@ -0,0 +1,4 @@ +include: _lithuanian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_lithuanian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Lithuanian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_include_base_44_malay.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_include_base_44_malay.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2928addefd54db40c79b4e99176081b4c64e6c1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_include_base_44_malay.yaml @@ -0,0 +1,10 @@ +group: include_base_44_malay +task: +- include_base_44_malay_few_shot_en_social_science +- include_base_44_malay_few_shot_en_business_commerce +- include_base_44_malay_few_shot_en_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_malay_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_malay_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd1a176785b9cdb5d0dcee947010f71d442a7423 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/_malay_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malay +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5f4ebecef53aba79e522a4d4de1c03773b32642 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _malay_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malay_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f12871347a8dd280ee96c9fa3a8822c6a0b1d14f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _malay_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_malay_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7dc708c657497682c7eaaabcb4f20e9a94976758 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/include_base_44_malay_social_science.yaml @@ -0,0 +1,5 @@ +include: _malay_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_malay_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malay/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_include_base_44_malayalam.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_include_base_44_malayalam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f90965f004519d9a27c69165748a2cea4d3955e6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_include_base_44_malayalam.yaml @@ -0,0 +1,13 @@ +group: include_base_44_malayalam +task: +- include_base_44_malayalam_few_shot_en_stem +- include_base_44_malayalam_few_shot_en_marine_license +- include_base_44_malayalam_few_shot_en_health_oriented_education +- include_base_44_malayalam_few_shot_en_arts_humanities +- include_base_44_malayalam_few_shot_en_social_science +- include_base_44_malayalam_few_shot_en_general_knowledge +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_malayalam_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_malayalam_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..62355b0098c9dbc9aca70e081214193483c1e044 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/_malayalam_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malayalam +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff25ab3c9d647d14e1032e15e2283ad670a79f01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malayalam_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0d396f3ca7480a72d41d551c132bd67904c0df0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_general_knowledge.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about General + knowledge. +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_malayalam_few_shot_en_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d98dcdeec2c4b370e8ca4b50df3419f1c0fdc2f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_malayalam_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_marine_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_marine_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..621fbe5609db4d05012c51b44afbca4df86bbb50 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_marine_license.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Marine + License. +process_docs: !function 'utils.process_marine_license' +task: include_base_44_malayalam_few_shot_en_marine_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..15c0bf408e21faa40f0443cdb2d08e7556cddfbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_social_science.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_malayalam_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f58e02ba6bb1645381a478a36963029d40d4c743 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/include_base_44_malayalam_stem.yaml @@ -0,0 +1,4 @@ +include: _malayalam_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_malayalam_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Malayalam/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_include_base_44_nepali.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_include_base_44_nepali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7203c2c0e837c548debed9c775d0f877378adf92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_include_base_44_nepali.yaml @@ -0,0 +1,9 @@ +group: include_base_44_nepali +task: +- include_base_44_nepali_few_shot_en_driving_license +- include_base_44_nepali_few_shot_en_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_nepali_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_nepali_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3dd04705a7c11f0a07c683f6a5a2a0149ba6961 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/_nepali_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Nepali +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9c8f28bfdcbbd554a01b49a9554441b347dc6f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_driving_license.yaml @@ -0,0 +1,5 @@ +include: _nepali_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_nepali_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..38eae547a50967c45c10fe5dd068e616ab4f45ae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/include_base_44_nepali_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _nepali_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_nepali_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Nepali/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_include_base_44_north macedonian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_include_base_44_north macedonian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1466346bcb1c7a50dfc2052b9c1dd9b8fd3f43c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_include_base_44_north macedonian.yaml @@ -0,0 +1,11 @@ +group: include_base_44_north macedonian +task: +- include_base_44_north macedonian_few_shot_en_arts_humanities +- include_base_44_north macedonian_few_shot_en_stem +- include_base_44_north macedonian_few_shot_en_business_commerce +- include_base_44_north macedonian_few_shot_en_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_north macedonian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_north macedonian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..70cf2a749efe48b164ac1be21af65bdef8fbe508 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/_north macedonian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: North Macedonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f69d7a40d358b0f7464bb34df727437537623454 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _north macedonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_north macedonian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9803bf4633cd5f346ee564a3626a4f7516b093ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _north macedonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_north macedonian_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3466060100d62dd8ac6d2a8afe3a21e7f8df2b3f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_social_science.yaml @@ -0,0 +1,5 @@ +include: _north macedonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_north macedonian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..124bc8d21dd2d98552da178598e8ade728caaf1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/include_base_44_north macedonian_stem.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_north macedonian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/North Macedonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_include_base_44_persian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_include_base_44_persian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bbe66e63ef65049cc7d07861c75082489943a8f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_include_base_44_persian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_persian +task: +- include_base_44_persian_few_shot_en_arts_humanities +- include_base_44_persian_few_shot_en_social_science +- include_base_44_persian_few_shot_en_professional_certification +- include_base_44_persian_few_shot_en_stem +- include_base_44_persian_few_shot_en_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_persian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_persian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4e4eec0811a1ef02b5002f39fa0f2cef387a975 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/_persian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Persian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dc13e0b0123ac5b02060286fc134a7e68ddee588 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_persian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af2beb6b94b5e7c264bed2f7a3da71b9037872fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_driving_license.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_persian_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c85b243d584d45848b6bfd53d7c8f891e7a8592b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_persian_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..688908863d867276d60a0eca4b6b1297937d72d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_social_science.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_persian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8129e769c9bc88dda76022eea01a59da2e61fdc2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/include_base_44_persian_stem.yaml @@ -0,0 +1,4 @@ +include: _persian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_persian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Persian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_include_base_44_polish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_include_base_44_polish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a456b9818305c6f116628d21249650d0b0d3be9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_include_base_44_polish.yaml @@ -0,0 +1,10 @@ +group: include_base_44_polish +task: +- include_base_44_polish_few_shot_en_professional_certification +- include_base_44_polish_few_shot_en_social_science +- include_base_44_polish_few_shot_en_stem +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_polish_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_polish_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..144716e8d8caa4ff6243c4b81a9daaadc4aa33fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/_polish_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Polish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9356930d52649940962eb6da8a3d8da1d7038c3e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _polish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Professional + certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_polish_few_shot_en_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76936c3949ffa052db43990516e5580fea7257cf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_social_science.yaml @@ -0,0 +1,5 @@ +include: _polish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_polish_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7a3a2d9bdfaa54571d99e38800a63cf047d50b64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/include_base_44_polish_stem.yaml @@ -0,0 +1,4 @@ +include: _polish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_polish_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Polish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_include_base_44_portuguese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_include_base_44_portuguese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7560d33aa3409a82111a19d8fb7e1d5e6aa83d23 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_include_base_44_portuguese.yaml @@ -0,0 +1,13 @@ +group: include_base_44_portuguese +task: +- include_base_44_portuguese_few_shot_en_stem +- include_base_44_portuguese_few_shot_en_social_science +- include_base_44_portuguese_few_shot_en_arts_humanities +- include_base_44_portuguese_few_shot_en_health_oriented_education +- include_base_44_portuguese_few_shot_en_business_commerce +- include_base_44_portuguese_few_shot_en_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_portuguese_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_portuguese_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ae882b22cd6f4a48c89737dfb7acc19b53add17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/_portuguese_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Portuguese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd78efab6fa8fc8385cbe3a9a2cbc39e9d018e12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_applied_science.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_portuguese_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16562b759af8aa50b8c0805108b00abb9d2a7625 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_portuguese_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11aeb1a4920daa4261fa8a876082e6f8428d9b6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_portuguese_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4f86708222b96e2771548769876bf1d43ed86fbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_portuguese_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e337f34c8fab084a5fb105c718fd33d95da35fdd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_social_science.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_portuguese_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c033afbf38a0672b8e63598dfa598c1e5417838 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/include_base_44_portuguese_stem.yaml @@ -0,0 +1,4 @@ +include: _portuguese_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_portuguese_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Portuguese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_include_base_44_russian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_include_base_44_russian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..12cb647237c1e581944795bcb3e3e78e6bdee5aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_include_base_44_russian.yaml @@ -0,0 +1,15 @@ +group: include_base_44_russian +task: +- include_base_44_russian_few_shot_en_stem +- include_base_44_russian_few_shot_en_health_oriented_education +- include_base_44_russian_few_shot_en_arts_humanities +- include_base_44_russian_few_shot_en_driving_license +- include_base_44_russian_few_shot_en_social_science +- include_base_44_russian_few_shot_en_business_commerce +- include_base_44_russian_few_shot_en_marine_license +- include_base_44_russian_few_shot_en_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_russian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_russian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b0e24871dfd767ebaa38747e9942f36c6222461 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/_russian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Russian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce4bd7525db332c187781697c5ef12ee851bd522 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_russian_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2106585d119961440ca310a3f85bad942eaf6207 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_russian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c2569ab7e7aa0a17a6bf080f29e68988592e948 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_russian_few_shot_en_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d4221fff78904c3249f4ae5c86f99826d44b38eb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_driving_license.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_russian_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d21eaf9e0b29a2c8ee5f5c66e0b1234e41871c29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_russian_few_shot_en_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_marine_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_marine_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee308374211737ddf429e44d44fcfc70be045022 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_marine_license.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Marine + License. +process_docs: !function 'utils.process_marine_license' +task: include_base_44_russian_few_shot_en_marine_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67f1f6c67fb632e5a22121b47499faa5ce43875a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_social_science.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_russian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5271751f5b7f106b96d436ece5bc95a30abd2d00 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/include_base_44_russian_stem.yaml @@ -0,0 +1,4 @@ +include: _russian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_russian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Russian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_include_base_44_serbian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_include_base_44_serbian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec156803da858e2a74b4c71feac47b357494499a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_include_base_44_serbian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_serbian +task: +- include_base_44_serbian_few_shot_en_stem +- include_base_44_serbian_few_shot_en_arts_humanities +- include_base_44_serbian_few_shot_en_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_serbian_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_serbian_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b413a08ad6dd70cf46a1272743175fc1d0153fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/_serbian_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Serbian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..af58929aed30314dceb89786b19a7d8e066af375 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _serbian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_serbian_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a6f8a6040cb14ab2aae57b7046b99091fea8be9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_social_science.yaml @@ -0,0 +1,5 @@ +include: _serbian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_serbian_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..723f01dec8b288e6391c2b3a93d5a96ed5a38002 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Serbian/include_base_44_serbian_stem.yaml @@ -0,0 +1,4 @@ +include: _serbian_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_serbian_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_include_base_44_spanish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_include_base_44_spanish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2fa8d6747572e2e949dc1482ebf1c082b86a118 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_include_base_44_spanish.yaml @@ -0,0 +1,11 @@ +group: include_base_44_spanish +task: +- include_base_44_spanish_few_shot_en_stem +- include_base_44_spanish_few_shot_en_social_science +- include_base_44_spanish_few_shot_en_arts_humanities +- include_base_44_spanish_few_shot_en_health_oriented_education +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_spanish_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_spanish_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..40b8981d6488dd5c326d16096713b769c2214dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/_spanish_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Spanish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23e1c959653ea87bfd8c73e869f75beb19de5806 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _spanish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_spanish_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3d2f1573ff743d40d9f4520dfdde6e0ae974e2b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Spanish/include_base_44_spanish_social_science.yaml @@ -0,0 +1,5 @@ +include: _spanish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_spanish_few_shot_en_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/_include_base_44_tagalog.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/_include_base_44_tagalog.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b33252de40c977e4ad83cb3f67035694cc4dfba5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/_include_base_44_tagalog.yaml @@ -0,0 +1,9 @@ +group: include_base_44_tagalog +task: +- include_base_44_tagalog_few_shot_en_arts_humanities +- include_base_44_tagalog_few_shot_en_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/include_base_44_tagalog_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/include_base_44_tagalog_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c346210a96082f7cfdf77335b0b8b17cd0307bd7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/include_base_44_tagalog_driving_license.yaml @@ -0,0 +1,5 @@ +include: _tagalog_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_tagalog_few_shot_en_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tagalog/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/_tamil_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/_tamil_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a659e689aab175e000cb61b4c6ee93743b18306c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/_tamil_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Tamil +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Tamil/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/_telugu_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/_telugu_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e6b1889ba37a23b0e7ff00d5176185bec2e33122 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/_telugu_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Telugu +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67307f47c6ee344e8106548e22cf6307a4839956 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_applied_science.yaml @@ -0,0 +1,5 @@ +include: _telugu_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_telugu_few_shot_en_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63a4805abde0a9fbe2e6edc7698e00fb59436857 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Telugu/include_base_44_telugu_stem.yaml @@ -0,0 +1,4 @@ +include: _telugu_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_telugu_few_shot_en_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/_turkish_few_shot_en_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/_turkish_few_shot_en_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..1f42709007eddc3b172eea6f017ff24a0181b78c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/_turkish_few_shot_en_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Turkish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/include_base_44_turkish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/include_base_44_turkish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c94da69a1654cc1ec08703ad96e9b3d7122b6f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_en/Turkish/include_base_44_turkish_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _turkish_few_shot_en_template_yaml +description: The following are multiple-choice questions (with answers) about Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_turkish_few_shot_en_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e03965ebd7615f7f4628feafbd3cdcb580cf4d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_general_knowledge.yaml @@ -0,0 +1,4 @@ +include: _bengali_few_shot_og_template_yaml +description: নিম্নলিখিতগুলি General knowledge সম্পর্কে বহু-পছন্দের প্রশ্ন (উত্তর সহ)। +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_bengali_few_shot_og_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0894175bbf527e7bf79d882c253a4979a563db3d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bengali/include_base_44_bengali_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _bengali_few_shot_og_template_yaml +description: নিম্নলিখিতগুলি Professional certification সম্পর্কে বহু-পছন্দের প্রশ্ন + (উত্তর সহ)। +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_bengali_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bulgarian/include_base_44_bulgarian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bulgarian/include_base_44_bulgarian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..23bfa235adb60b8f61d31df262846587f26aa1d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Bulgarian/include_base_44_bulgarian_social_science.yaml @@ -0,0 +1,4 @@ +include: _bulgarian_few_shot_og_template_yaml +description: Следват въпроси с избираем отговор (с отговори) за Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_bulgarian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/_chinese_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/_chinese_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..17d8f3ebfcbae471a4c7d4a852efb489be4bbd18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/_chinese_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Chinese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2ecb579d791ee19b9729adec85c7dbc4467e3de --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_applied_science.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Applied Science 的多项选择题(附答案)。 +process_docs: !function 'utils.process_applied_science' +task: include_base_44_chinese_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..667d3766653b02cd21eea2309e32cea6fc556178 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Arts & Humanities 的多项选择题(附答案)。 +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_chinese_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..272b5e6728125a7a3a13aca9df290cd6e1ffa3c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_business_commerce.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Business & Commerce 的多项选择题(附答案)。 +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_chinese_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cac0134f55897803a6c00de52301eb31764b6fc0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_driving_license.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Driving License 的多项选择题(附答案)。 +process_docs: !function 'utils.process_driving_license' +task: include_base_44_chinese_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..438aed8a87351a9e94bb498ca42516c508431d1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_health_oriented_education.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Health oriented education 的多项选择题(附答案)。 +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_chinese_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5632cd692ed867332c30dd246129d65f4e3910c7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_social_science.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 Social Science 的多项选择题(附答案)。 +process_docs: !function 'utils.process_social_science' +task: include_base_44_chinese_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..525ec63d852214ffef6073d926255de43e28051d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/include_base_44_chinese_stem.yaml @@ -0,0 +1,4 @@ +include: _chinese_few_shot_og_template_yaml +description: 以下是关于 STEM 的多项选择题(附答案)。 +process_docs: !function 'utils.process_stem' +task: include_base_44_chinese_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Chinese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_croatian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_croatian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..97712c7f37bbc604dd505735e3e3000610527ba0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_croatian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Croatian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_include_base_44_croatian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_include_base_44_croatian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b278be8b0174d91c5ccef792c551b0e8bd645b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/_include_base_44_croatian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_croatian +task: +- include_base_44_croatian_few_shot_og_stem +- include_base_44_croatian_few_shot_og_arts_humanities +- include_base_44_croatian_few_shot_og_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..844a964802a1532cd23e931cf0327550145b5351 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _croatian_few_shot_og_template_yaml +description: Slijede pitanja s višestrukim izborom (s odgovorima) o Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_croatian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73bba5f39dc2fb47c61b118d043276fc8790dfdf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_social_science.yaml @@ -0,0 +1,4 @@ +include: _croatian_few_shot_og_template_yaml +description: Slijede pitanja s višestrukim izborom (s odgovorima) o Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_croatian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..32dad4ce3a7367ed640ea69b489f8062c986760a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Croatian/include_base_44_croatian_stem.yaml @@ -0,0 +1,4 @@ +include: _croatian_few_shot_og_template_yaml +description: Slijede pitanja s višestrukim izborom (s odgovorima) o STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_croatian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_dutch_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_dutch_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..6aebdef762b40d9af6da20a14cf882c40c2790e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_dutch_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Dutch +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_include_base_44_dutch.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_include_base_44_dutch.yaml new file mode 100644 index 0000000000000000000000000000000000000000..038431e4d76824fbb791dcf05419faa7148d1bca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/_include_base_44_dutch.yaml @@ -0,0 +1,12 @@ +group: include_base_44_dutch +task: +- include_base_44_dutch_few_shot_og_arts_humanities +- include_base_44_dutch_few_shot_og_stem +- include_base_44_dutch_few_shot_og_social_science +- include_base_44_dutch_few_shot_og_health_oriented_education +- include_base_44_dutch_few_shot_og_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7835e016efca8d5ce0d1c0b0037e06c05263828c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_applied_science.yaml @@ -0,0 +1,4 @@ +include: _dutch_few_shot_og_template_yaml +description: Hieronder staan ​​meerkeuzevragen (met antwoorden) over Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_dutch_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6e0286c50950f2b0ac5fc445f89160126b7d7a18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _dutch_few_shot_og_template_yaml +description: Hieronder staan ​​meerkeuzevragen (met antwoorden) over Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_dutch_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..66b7b91195c54a979cd14fb7037a14b08eac7562 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _dutch_few_shot_og_template_yaml +description: Hieronder staan ​​meerkeuzevragen (met antwoorden) over Health oriented + education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_dutch_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2cc5cab13b5487e68a4d3214b1830d266ff9b1af --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_social_science.yaml @@ -0,0 +1,4 @@ +include: _dutch_few_shot_og_template_yaml +description: Hieronder staan ​​meerkeuzevragen (met antwoorden) over Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_dutch_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..81051d20ae7409df4d80d82aa66a13eb98cc1a35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/include_base_44_dutch_stem.yaml @@ -0,0 +1,4 @@ +include: _dutch_few_shot_og_template_yaml +description: Hieronder staan ​​meerkeuzevragen (met antwoorden) over STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_dutch_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Dutch/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_estonian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_estonian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8d1a6246ce099a6bfe4940924f41d3614bfbb47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_estonian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Estonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_include_base_44_estonian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_include_base_44_estonian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e162ba249a9b19bb6795764857c10bc299dd8dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/_include_base_44_estonian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_estonian +task: +- include_base_44_estonian_few_shot_og_health_oriented_education +- include_base_44_estonian_few_shot_og_social_science +- include_base_44_estonian_few_shot_og_applied_science +- include_base_44_estonian_few_shot_og_stem +- include_base_44_estonian_few_shot_og_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7114df96f354b43977ac6f12f37293432f1991d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_og_template_yaml +description: Järgmised on valikvastustega küsimused (koos vastustega) teemal Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_estonian_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80e0ef6fc41429f6cd1f513c1b701bac67545c47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_og_template_yaml +description: Järgmised on valikvastustega küsimused (koos vastustega) teemal Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_estonian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7136e8ab6c857d0df1744c5bfbf31ec0f571a89a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_og_template_yaml +description: Järgmised on valikvastustega küsimused (koos vastustega) teemal Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_estonian_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c60fdd71e119fa954c1da1e1622ed5d60bf66076 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_social_science.yaml @@ -0,0 +1,5 @@ +include: _estonian_few_shot_og_template_yaml +description: Järgmised on valikvastustega küsimused (koos vastustega) teemal Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_estonian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b5a26d3e9d514142a9ec3f93557b2c1097c0041c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/include_base_44_estonian_stem.yaml @@ -0,0 +1,4 @@ +include: _estonian_few_shot_og_template_yaml +description: Järgmised on valikvastustega küsimused (koos vastustega) teemal STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_estonian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Estonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_finnish_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_finnish_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce7cc1f645feee31a8cf0472e65894127df83315 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_finnish_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Finnish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_include_base_44_finnish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_include_base_44_finnish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6a14464c43b04271a3ad7f01a72fa22a981473fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/_include_base_44_finnish.yaml @@ -0,0 +1,12 @@ +group: include_base_44_finnish +task: +- include_base_44_finnish_few_shot_og_stem +- include_base_44_finnish_few_shot_og_arts_humanities +- include_base_44_finnish_few_shot_og_social_science +- include_base_44_finnish_few_shot_og_health_oriented_education +- include_base_44_finnish_few_shot_og_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65111785c759d574e8b5fe5bb3da56130c95b722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_og_template_yaml +description: Seuraavat ovat monivalintakysymyksiä (vastauksineen) aiheesta Arts & + Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_finnish_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d15459cf57b9e987a765edba641e39324dfe5fb1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_og_template_yaml +description: Seuraavat ovat monivalintakysymyksiä (vastauksineen) aiheesta Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_finnish_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..928f1d59004d968c6ee9f7be410bd48cf9b89649 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_social_science.yaml @@ -0,0 +1,5 @@ +include: _finnish_few_shot_og_template_yaml +description: Seuraavat ovat monivalintakysymyksiä (vastauksineen) aiheesta Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_finnish_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dbe796931000a60404952c77a1f0ec56fe85670e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/include_base_44_finnish_stem.yaml @@ -0,0 +1,4 @@ +include: _finnish_few_shot_og_template_yaml +description: Seuraavat ovat monivalintakysymyksiä (vastauksineen) aiheesta STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_finnish_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Finnish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_french_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_french_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..c11e99be2f82824aa12c56f665f8a66a707a4539 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_french_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: French +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_include_base_44_french.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_include_base_44_french.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6670e56a13d4c2f04ef2eadbfe04f54e6e02d948 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/_include_base_44_french.yaml @@ -0,0 +1,12 @@ +group: include_base_44_french +task: +- include_base_44_french_few_shot_og_stem +- include_base_44_french_few_shot_og_social_science +- include_base_44_french_few_shot_og_health_oriented_education +- include_base_44_french_few_shot_og_arts_humanities +- include_base_44_french_few_shot_og_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..97485b4701eda6036ff5f4f1f1f626a700a115a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_og_template_yaml +description: Les questions suivantes sont à choix multiples (avec réponses) sur Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_french_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9cd14fda68bf52c07a8883cc277adc73462342cb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_driving_license.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_og_template_yaml +description: Les questions suivantes sont à choix multiples (avec réponses) sur Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_french_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7f1bced8ccc8a805bfe9eda7096ad1256be515b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_og_template_yaml +description: Les questions suivantes sont à choix multiples (avec réponses) sur Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_french_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac980e78f852cffdff19ec4d75ed8940c6995ca3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_social_science.yaml @@ -0,0 +1,5 @@ +include: _french_few_shot_og_template_yaml +description: Les questions suivantes sont à choix multiples (avec réponses) sur Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_french_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bee92ec21f70f62fb85c5f7cdcd6f8900f7e7863 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/include_base_44_french_stem.yaml @@ -0,0 +1,4 @@ +include: _french_few_shot_og_template_yaml +description: Les questions suivantes sont à choix multiples (avec réponses) sur STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_french_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/French/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_georgian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_georgian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..e827d49433a43f554715863cf8e3cd396be857ce --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_georgian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Georgian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_include_base_44_georgian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_include_base_44_georgian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7237632a69107b1169fff808b0d0c5a0f55f4135 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/_include_base_44_georgian.yaml @@ -0,0 +1,8 @@ +group: include_base_44_georgian +task: +- include_base_44_georgian_few_shot_og_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/include_base_44_georgian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/include_base_44_georgian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8886a9952e0e9ef57e76bc84a57fd84168f15463 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/include_base_44_georgian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _georgian_few_shot_og_template_yaml +description: ქვემოთ მოცემულია რამდენიმე არჩევანის კითხვები (პასუხებით) Arts & Humanities-ის + შესახებ. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_georgian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Georgian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_german_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_german_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..914f2ab8c77424fb8b615467278296dad47dd6c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_german_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: German +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_include_base_44_german.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_include_base_44_german.yaml new file mode 100644 index 0000000000000000000000000000000000000000..68a60ced72613ef79558f10bc7cbb43fd05e8003 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/_include_base_44_german.yaml @@ -0,0 +1,10 @@ +group: include_base_44_german +task: +- include_base_44_german_few_shot_og_stem +- include_base_44_german_few_shot_og_social_science +- include_base_44_german_few_shot_og_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4edd5e30161c36431ec4533c28133f8ad4b6e73c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_driving_license.yaml @@ -0,0 +1,5 @@ +include: _german_few_shot_og_template_yaml +description: Nachfolgend finden Sie Multiple-Choice-Fragen (mit Antworten) zu Driving + License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_german_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..70ae02e297895eeeed0b7bcb1870e542370a96f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_social_science.yaml @@ -0,0 +1,5 @@ +include: _german_few_shot_og_template_yaml +description: Nachfolgend finden Sie Multiple-Choice-Fragen (mit Antworten) zu Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_german_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c92262301c81e66bd912c0f36d15eb33bb82c74a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/include_base_44_german_stem.yaml @@ -0,0 +1,4 @@ +include: _german_few_shot_og_template_yaml +description: Nachfolgend finden Sie Multiple-Choice-Fragen (mit Antworten) zu STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_german_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/German/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_greek_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_greek_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7a822fc2f49ccd11d801864a434b5d23675e8b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_greek_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Greek +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_include_base_44_greek.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_include_base_44_greek.yaml new file mode 100644 index 0000000000000000000000000000000000000000..efd7f55013af8518082043618a6928ee4bc850d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/_include_base_44_greek.yaml @@ -0,0 +1,14 @@ +group: include_base_44_greek +task: +- include_base_44_greek_few_shot_og_stem +- include_base_44_greek_few_shot_og_arts_humanities +- include_base_44_greek_few_shot_og_social_science +- include_base_44_greek_few_shot_og_business_commerce +- include_base_44_greek_few_shot_og_health_oriented_education +- include_base_44_greek_few_shot_og_professional_certification +- include_base_44_greek_few_shot_og_medical_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..79005600bd06953dde70dfa9ec8be5fed21ab22d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_greek_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c83b39bfa13c165e5deb4ed6299badb4d6ddd1ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_greek_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bac35ebb8f3a6e5d65856f601d116e827dfb82f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_greek_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27266d5051e727cbaee057a639cf5b60089d6ea8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_medical_license.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Medical License. +process_docs: !function 'utils.process_medical_license' +task: include_base_44_greek_few_shot_og_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..59d378c3f839c7f4227a0b97c04c55dea6433183 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_greek_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..698c2346b1e2f97ed3e8dd9251446669237930b7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_social_science.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_greek_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c0e83f200f4ae65ed29a113e1c147854fbcd35b6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/include_base_44_greek_stem.yaml @@ -0,0 +1,5 @@ +include: _greek_few_shot_og_template_yaml +description: Ακολουθούν ερωτήσεις πολλαπλής επιλογής (με απαντήσεις) σχετικά με το + STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_greek_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Greek/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_hebrew_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_hebrew_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fc6fcfa2e4ad9555b382ab58009a847a42e2cf1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_hebrew_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hebrew +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_include_base_44_hebrew.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_include_base_44_hebrew.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b547679ff9fc7ac7aed08c25f85ef17a4eb0a10 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/_include_base_44_hebrew.yaml @@ -0,0 +1,9 @@ +group: include_base_44_hebrew +task: +- include_base_44_hebrew_few_shot_og_arts_humanities +- include_base_44_hebrew_few_shot_og_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d7c19fd6faf2733a01d7e89215bef2506dd796e0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _hebrew_few_shot_og_template_yaml +description: להלן שאלות ברירות רבות (עם תשובות) על Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hebrew_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ff9785b3cadc55298839cc72f3acfac3c69f34d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/include_base_44_hebrew_driving_license.yaml @@ -0,0 +1,4 @@ +include: _hebrew_few_shot_og_template_yaml +description: להלן שאלות ברירות רבות (עם תשובות) על Driving license. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hebrew_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hebrew/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_hindi_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_hindi_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..bf6ca8c1e94d002ebb64fec867d30694ff2efc81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_hindi_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hindi +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_include_base_44_hindi.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_include_base_44_hindi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d700e8beab55f18fdb04c9ad9329ac4c94d6ad81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/_include_base_44_hindi.yaml @@ -0,0 +1,15 @@ +group: include_base_44_hindi +task: +- include_base_44_hindi_few_shot_og_professional_certification +- include_base_44_hindi_few_shot_og_stem +- include_base_44_hindi_few_shot_og_social_science +- include_base_44_hindi_few_shot_og_driving_license +- include_base_44_hindi_few_shot_og_applied_science +- include_base_44_hindi_few_shot_og_arts_humanities +- include_base_44_hindi_few_shot_og_general_knowledge +- include_base_44_hindi_few_shot_og_health_oriented_education +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2984805a4e5e25f3633f567d629ca8609d31a92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_applied_science.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Applied Science के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) + हैं। +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hindi_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cee13069f8e05cb100a9c8cf360dea18a2611057 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Arts & Humanities के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) + हैं। +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_hindi_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9bc2e52d50b662b1e79e27bd48b8170314c72a1a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_driving_license.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Driving License के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) + हैं। +process_docs: !function 'utils.process_driving_license' +task: include_base_44_hindi_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8467fa94a7ee76a1d2e6a9316818606bed633a6d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_general_knowledge.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित General knowledge के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) + हैं। +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_hindi_few_shot_og_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3651f6aa3eec2a518c9ac67cf2b69a65b1ee2f62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Health oriented education के बारे में बहुविकल्पीय प्रश्न (उत्तर + सहित) हैं। +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_hindi_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4e81ef6e8ed39ecb5d53fddf16112f16ad507163 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Professional certification के बारे में बहुविकल्पीय प्रश्न + (उत्तर सहित) हैं। +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_hindi_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bcd4dfd1621e286c4d47616c991fc9f8fc123415 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_social_science.yaml @@ -0,0 +1,5 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित Social Science के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) + हैं। +process_docs: !function 'utils.process_social_science' +task: include_base_44_hindi_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..efe318ef7ec7b2650929809ce0f1c4ee0f43d2ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/include_base_44_hindi_stem.yaml @@ -0,0 +1,4 @@ +include: _hindi_few_shot_og_template_yaml +description: निम्नलिखित STEM के बारे में बहुविकल्पीय प्रश्न (उत्तर सहित) हैं। +process_docs: !function 'utils.process_stem' +task: include_base_44_hindi_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hindi/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_hungarian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_hungarian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..47b30358909e8a9b3e2097202db96573634f49d7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_hungarian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Hungarian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_include_base_44_hungarian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_include_base_44_hungarian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..768eacd13fcd40359780afeab3f42d341ad65515 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/_include_base_44_hungarian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_hungarian +task: +- include_base_44_hungarian_few_shot_og_stem +- include_base_44_hungarian_few_shot_og_applied_science +- include_base_44_hungarian_few_shot_og_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ddcd1081dfea187136229c2044dce4bde76f5bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _hungarian_few_shot_og_template_yaml +description: 'Az alábbiak feleletválasztós kérdések (válaszokkal) a következővel kapcsolatban: + Applied Science.' +process_docs: !function 'utils.process_applied_science' +task: include_base_44_hungarian_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cbc4cc7995c64adcb741a1986048a692b4560402 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_social_science.yaml @@ -0,0 +1,5 @@ +include: _hungarian_few_shot_og_template_yaml +description: 'Az alábbiak feleletválasztós kérdések (válaszokkal) a következővel kapcsolatban: + Social Science.' +process_docs: !function 'utils.process_social_science' +task: include_base_44_hungarian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20b80c910ed81c48e381dcaf405c9fb425311db6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/include_base_44_hungarian_stem.yaml @@ -0,0 +1,5 @@ +include: _hungarian_few_shot_og_template_yaml +description: 'Az alábbiak feleletválasztós kérdések (válaszokkal) a következővel kapcsolatban: + STEM.' +process_docs: !function 'utils.process_stem' +task: include_base_44_hungarian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Hungarian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_include_base_44_indonesian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_include_base_44_indonesian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b15de5264af579267c5958c99baf1462d7e28c62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_include_base_44_indonesian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_indonesian +task: +- include_base_44_indonesian_few_shot_og_arts_humanities +- include_base_44_indonesian_few_shot_og_social_science +- include_base_44_indonesian_few_shot_og_stem +- include_base_44_indonesian_few_shot_og_applied_science +- include_base_44_indonesian_few_shot_og_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_indonesian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_indonesian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..eed3a59c1b3ff5717e07aad4acd3ab6a4a1fce23 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/_indonesian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Indonesian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7d76b9e215cace4a79cb2934087767466f911363 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_og_template_yaml +description: Berikut ini adalah pertanyaan pilihan ganda (dengan jawaban) tentang + Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_indonesian_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10c4f24139c54c9798259f69fd8de7d32ed52bc3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_og_template_yaml +description: Berikut ini adalah pertanyaan pilihan ganda (dengan jawaban) tentang + Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_indonesian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2a526c8a0352d22ce0afbb9bc7a6afdf09c5a4c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_og_template_yaml +description: Berikut ini adalah pertanyaan pilihan ganda (dengan jawaban) tentang + Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_indonesian_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a473b48e3938f8b6aa434ec3063f078ede94b9bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_social_science.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_og_template_yaml +description: Berikut ini adalah pertanyaan pilihan ganda (dengan jawaban) tentang + Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_indonesian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce374b765c94668847438454aa08223b8bd156a3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/include_base_44_indonesian_stem.yaml @@ -0,0 +1,5 @@ +include: _indonesian_few_shot_og_template_yaml +description: Berikut ini adalah pertanyaan pilihan ganda (dengan jawaban) tentang + STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_indonesian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Indonesian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_include_base_44_italian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_include_base_44_italian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16390f45f8d99a4a4f785b0dfdf01e1bd98b799a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_include_base_44_italian.yaml @@ -0,0 +1,13 @@ +group: include_base_44_italian +task: +- include_base_44_italian_few_shot_og_stem +- include_base_44_italian_few_shot_og_arts_humanities +- include_base_44_italian_few_shot_og_social_science +- include_base_44_italian_few_shot_og_applied_science +- include_base_44_italian_few_shot_og_health_oriented_education +- include_base_44_italian_few_shot_og_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_italian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_italian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8bd03586e0eb9a4e3cf4121af6e409df444ff88 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/_italian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Italian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1c3f625c7346cb74677bc84bfc1a016ac72066b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_italian_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b416ed9d4bf4002668ff1f2767a1fba5de7f45ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_italian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb6f0dddda4c38a8f64cd960a58e347b12a7340d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_italian_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b01093fbbad9ae830ec78d74796e0d68ef941b98 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_italian_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dfa443d5ebfd99f21e9da7370e818f8d24332907 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_social_science.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_italian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3431a9872cd59a295de70a928bc72c42ad0ea41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/include_base_44_italian_stem.yaml @@ -0,0 +1,5 @@ +include: _italian_few_shot_og_template_yaml +description: Di seguito sono riportate domande a scelta multipla (con risposte) su + STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_italian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Italian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_include_base_44_japanese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_include_base_44_japanese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e72c5bb432b0e81de637702186d1a7f90521071c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_include_base_44_japanese.yaml @@ -0,0 +1,10 @@ +group: include_base_44_japanese +task: +- include_base_44_japanese_few_shot_og_driving_license +- include_base_44_japanese_few_shot_og_medical_license +- include_base_44_japanese_few_shot_og_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_japanese_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_japanese_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..878d1a96323249d329613c541aa3326c5e948c16 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/_japanese_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Japanese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8ac3e66d6e035f5846385e30585f09db6d6e0dec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_driving_license.yaml @@ -0,0 +1,4 @@ +include: _japanese_few_shot_og_template_yaml +description: 以下は、Driving License に関する複数選択の質問(回答付き)です。 +process_docs: !function 'utils.process_driving_license' +task: include_base_44_japanese_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_medical_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_medical_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49e9bd106f5b439ecd64af97b5640dfbd57c6064 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_medical_license.yaml @@ -0,0 +1,4 @@ +include: _japanese_few_shot_og_template_yaml +description: 以下は、Medical License に関する複数選択の質問(回答付き)です。 +process_docs: !function 'utils.process_medical_license' +task: include_base_44_japanese_few_shot_og_medical_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5e4d99df24c73b73ffadd3c4d5d0a527173fee43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/include_base_44_japanese_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _japanese_few_shot_og_template_yaml +description: 以下は、Professional certification に関する複数選択の質問(回答付き)です。 +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_japanese_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Japanese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_include_base_44_kazakh.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_include_base_44_kazakh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4d7be337dc5bde99d7f08dd9f5e528cb70b8aa9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_include_base_44_kazakh.yaml @@ -0,0 +1,8 @@ +group: include_base_44_kazakh +task: +- include_base_44_kazakh_few_shot_og_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_kazakh_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_kazakh_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a800eeb37b1675b576a69cbc3267c9e3c9cc86bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/_kazakh_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Kazakh +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/include_base_44_kazakh_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/include_base_44_kazakh_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b7ebf10cc509eb81f6b69f09552783804e83bdd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/include_base_44_kazakh_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _kazakh_few_shot_og_template_yaml +description: Төменде Arts & Humanities туралы бірнеше таңдаулы сұрақтар (жауаптары + бар) берілген. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_kazakh_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Kazakh/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_include_base_44_korean.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_include_base_44_korean.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0794fcf4802f63bf9f056e121b1bc231a8ee768a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_include_base_44_korean.yaml @@ -0,0 +1,9 @@ +group: include_base_44_korean +task: +- include_base_44_korean_few_shot_og_professional_certification +- include_base_44_korean_few_shot_og_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_korean_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_korean_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd5d2e07e2baea8f60c0b3ab9d9c80aeaf2ba8d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/_korean_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Korean +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..064913e01ec5be601936529cff7fc58c25b077fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_professional_certification.yaml @@ -0,0 +1,4 @@ +include: _korean_few_shot_og_template_yaml +description: 다음은 Professional certification에 대한 객관식 질문(답변 포함)입니다. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_korean_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3184c98192a5a52af0f345cdc636e33e6c701f43 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/include_base_44_korean_social_science.yaml @@ -0,0 +1,4 @@ +include: _korean_few_shot_og_template_yaml +description: 다음은 Social Science에 대한 객관식 질문(답변 포함)입니다. +process_docs: !function 'utils.process_social_science' +task: include_base_44_korean_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Korean/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_include_base_44_lithuanian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_include_base_44_lithuanian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dcc60df6b682c24b00641a215bafac7fc3c0477a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_include_base_44_lithuanian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_lithuanian +task: +- include_base_44_lithuanian_few_shot_og_arts_humanities +- include_base_44_lithuanian_few_shot_og_stem +- include_base_44_lithuanian_few_shot_og_social_science +- include_base_44_lithuanian_few_shot_og_business_commerce +- include_base_44_lithuanian_few_shot_og_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_lithuanian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_lithuanian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..64315ee76457e6fa7e94c34e9d27be8808c3e1a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/_lithuanian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Lithuanian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3a1a819578a45e3fe2184af89cb75dca6409cc64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_og_template_yaml +description: Toliau pateikiami klausimai su atsakymų variantais (su atsakymais) apie + Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_lithuanian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..edf2f1df9fcf89af408ba364aa592fc7ed3d106b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_og_template_yaml +description: Toliau pateikiami klausimai su atsakymų variantais (su atsakymais) apie + Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_lithuanian_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58a38b95132b06993379f4471f8748145d18edcd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_og_template_yaml +description: Toliau pateikiami klausimai su atsakymų variantais (su atsakymais) apie + Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_lithuanian_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdcc0e48f4504e4cd7cc75a4fa3aff1aa64250ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_social_science.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_og_template_yaml +description: Toliau pateikiami klausimai su atsakymų variantais (su atsakymais) apie + Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_lithuanian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dfb4de1c1ec597e6f5de67779f0227292e6747c2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/include_base_44_lithuanian_stem.yaml @@ -0,0 +1,5 @@ +include: _lithuanian_few_shot_og_template_yaml +description: Toliau pateikiami klausimai su atsakymų variantais (su atsakymais) apie + STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_lithuanian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Lithuanian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_include_base_44_malay.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_include_base_44_malay.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dd79702020119c67aa228348a3a38631da9e2df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_include_base_44_malay.yaml @@ -0,0 +1,10 @@ +group: include_base_44_malay +task: +- include_base_44_malay_few_shot_og_social_science +- include_base_44_malay_few_shot_og_business_commerce +- include_base_44_malay_few_shot_og_arts_humanities +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_malay_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_malay_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..fd1a176785b9cdb5d0dcee947010f71d442a7423 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/_malay_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malay +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc393f94cc752bf2f2269e630e957b51a0dff3b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _malay_few_shot_og_template_yaml +description: Berikut ialah soalan aneka pilihan (dengan jawapan) tentang Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malay_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..92f74fe416d2b4a23d1bc4ece914f3d62a74c7aa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _malay_few_shot_og_template_yaml +description: Berikut ialah soalan aneka pilihan (dengan jawapan) tentang Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_malay_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..73be07ecaeb4c9fcca7a59749f65e97b5dec53ac --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/include_base_44_malay_social_science.yaml @@ -0,0 +1,4 @@ +include: _malay_few_shot_og_template_yaml +description: Berikut ialah soalan aneka pilihan (dengan jawapan) tentang Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_malay_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malay/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_include_base_44_malayalam.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_include_base_44_malayalam.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee2c106b34af9235fa9867419ab0350d6fbfa836 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_include_base_44_malayalam.yaml @@ -0,0 +1,13 @@ +group: include_base_44_malayalam +task: +- include_base_44_malayalam_few_shot_og_stem +- include_base_44_malayalam_few_shot_og_marine_license +- include_base_44_malayalam_few_shot_og_health_oriented_education +- include_base_44_malayalam_few_shot_og_arts_humanities +- include_base_44_malayalam_few_shot_og_social_science +- include_base_44_malayalam_few_shot_og_general_knowledge +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_malayalam_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_malayalam_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..62355b0098c9dbc9aca70e081214193483c1e044 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/_malayalam_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Malayalam +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a5512a18b5dd665fb3a3e50212e1c0c37ec36961 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ Arts & Humanities നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് ചോദ്യങ്ങളാണ് + (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_malayalam_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_general_knowledge.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_general_knowledge.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aa834c73d84151f27a1b056ae7dcf0dcf9597b07 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_general_knowledge.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ General knowledge നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് ചോദ്യങ്ങളാണ് + (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_general_knowledge' +task: include_base_44_malayalam_few_shot_og_general_knowledge diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4acabdb7eaf70b8037f592af15ea2205b1eac1fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ Health oriented education നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് + ചോദ്യങ്ങളാണ് (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_malayalam_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_marine_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_marine_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dae2239f25248586f580240606424eb4bb9b044a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_marine_license.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ Marine License നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് ചോദ്യങ്ങളാണ് + (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_marine_license' +task: include_base_44_malayalam_few_shot_og_marine_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1cb485f6170ad863a08467a0b952fd4ed8315728 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_social_science.yaml @@ -0,0 +1,5 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ Social Science നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് ചോദ്യങ്ങളാണ് + (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_social_science' +task: include_base_44_malayalam_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b3b9da9ce9f55ef604ed9a19199f0860ee42822 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/include_base_44_malayalam_stem.yaml @@ -0,0 +1,4 @@ +include: _malayalam_few_shot_og_template_yaml +description: ഇനിപ്പറയുന്നവ STEM നെക്കുറിച്ചുള്ള മൾട്ടിപ്പിൾ ചോയ്‌സ് ചോദ്യങ്ങളാണ് (ഉത്തരങ്ങളോടെ). +process_docs: !function 'utils.process_stem' +task: include_base_44_malayalam_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Malayalam/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_include_base_44_nepali.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_include_base_44_nepali.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7059a58557a11ffe2d3e97e101c8bce26b8c9d81 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_include_base_44_nepali.yaml @@ -0,0 +1,9 @@ +group: include_base_44_nepali +task: +- include_base_44_nepali_few_shot_og_driving_license +- include_base_44_nepali_few_shot_og_professional_certification +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_nepali_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_nepali_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a3dd04705a7c11f0a07c683f6a5a2a0149ba6961 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/_nepali_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Nepali +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9fb8a1fae82b7d400f940bdeca0aa15f81719280 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_driving_license.yaml @@ -0,0 +1,4 @@ +include: _nepali_few_shot_og_template_yaml +description: निम्न Driving License को बारेमा बहु-छनौट प्रश्नहरू (उत्तरहरू सहित) छन्। +process_docs: !function 'utils.process_driving_license' +task: include_base_44_nepali_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e2484a1e06313aa4b31de8b71544ce9c14a43934 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/include_base_44_nepali_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _nepali_few_shot_og_template_yaml +description: निम्न Professional certification को बारेमा बहु-छनौट प्रश्नहरू (उत्तरहरू + सहित) छन्। +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_nepali_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Nepali/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_include_base_44_north macedonian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_include_base_44_north macedonian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..78602e840edea664e1c244df938343c604d4204d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_include_base_44_north macedonian.yaml @@ -0,0 +1,11 @@ +group: include_base_44_north macedonian +task: +- include_base_44_north macedonian_few_shot_og_arts_humanities +- include_base_44_north macedonian_few_shot_og_stem +- include_base_44_north macedonian_few_shot_og_business_commerce +- include_base_44_north macedonian_few_shot_og_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_north macedonian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_north macedonian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..70cf2a749efe48b164ac1be21af65bdef8fbe508 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/_north macedonian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: North Macedonian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..416343952817778d7c4e956a6e7339c87a56fd2e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_arts_humanities.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_few_shot_og_template_yaml +description: Следниве се прашања со повеќекратен избор (со одговори) за Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_north macedonian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5c09e2453fd133ee6166bde6f2fa236ab4d87f11 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _north macedonian_few_shot_og_template_yaml +description: Следниве се прашања со повеќекратен избор (со одговори) за Business & + Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_north macedonian_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3afbb3b5fff477a9db05c618dabc09bf07cc7f5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_social_science.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_few_shot_og_template_yaml +description: Следниве се прашања со повеќекратен избор (со одговори) за Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_north macedonian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7f326a34db24fe6172bfa699e010030a840d098b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/include_base_44_north macedonian_stem.yaml @@ -0,0 +1,4 @@ +include: _north macedonian_few_shot_og_template_yaml +description: Следниве се прашања со повеќекратен избор (со одговори) за STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_north macedonian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/North Macedonian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_include_base_44_persian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_include_base_44_persian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6be6c1fd5be3aa5db39f2600b681ef05f623a187 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_include_base_44_persian.yaml @@ -0,0 +1,12 @@ +group: include_base_44_persian +task: +- include_base_44_persian_few_shot_og_arts_humanities +- include_base_44_persian_few_shot_og_social_science +- include_base_44_persian_few_shot_og_professional_certification +- include_base_44_persian_few_shot_og_stem +- include_base_44_persian_few_shot_og_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_persian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_persian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4e4eec0811a1ef02b5002f39fa0f2cef387a975 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/_persian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Persian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4af0fa628ec9520f338957178b45f6624263e51 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_og_template_yaml +description: در زیر سؤالات چند گزینه ای (همراه با پاسخ) در مورد Arts & Humanities + آمده است. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_persian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c4fd05f9de212135a132a4c2982de4b5a9e76c06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_driving_license.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_og_template_yaml +description: در زیر سؤالات چند گزینه ای (همراه با پاسخ) در مورد Driving License آمده + است. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_persian_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..16e9f72f70dd519a2da880573654cf3801f78356 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_og_template_yaml +description: در زیر سؤالات چند گزینه ای (همراه با پاسخ) در مورد Professional certification + آمده است. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_persian_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..630e0b2298157f3072dc706e74362ee128513a41 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_social_science.yaml @@ -0,0 +1,5 @@ +include: _persian_few_shot_og_template_yaml +description: در زیر سؤالات چند گزینه ای (همراه با پاسخ) در مورد Social Science آمده + است. +process_docs: !function 'utils.process_social_science' +task: include_base_44_persian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d1e9a130ce72fa0f6fb6fec87e07cf72ef21b7fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/include_base_44_persian_stem.yaml @@ -0,0 +1,4 @@ +include: _persian_few_shot_og_template_yaml +description: در زیر سؤالات چند گزینه ای (همراه با پاسخ) در مورد STEM آمده است. +process_docs: !function 'utils.process_stem' +task: include_base_44_persian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Persian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_include_base_44_polish.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_include_base_44_polish.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fbe7fe509bd142e6e1f510c4708695ede14bd29 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_include_base_44_polish.yaml @@ -0,0 +1,10 @@ +group: include_base_44_polish +task: +- include_base_44_polish_few_shot_og_professional_certification +- include_base_44_polish_few_shot_og_social_science +- include_base_44_polish_few_shot_og_stem +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_polish_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_polish_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..144716e8d8caa4ff6243c4b81a9daaadc4aa33fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/_polish_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Polish +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_professional_certification.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_professional_certification.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c46f3f8bbbfc11ca4a3ad7497d407167448c7b7e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_professional_certification.yaml @@ -0,0 +1,5 @@ +include: _polish_few_shot_og_template_yaml +description: Poniżej znajdują się pytania wielokrotnego wyboru (z odpowiedziami) na + temat Professional certification. +process_docs: !function 'utils.process_professional_certification' +task: include_base_44_polish_few_shot_og_professional_certification diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52e3ce5a8e7e26ef03028baaa818bbb87ded1ee5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_social_science.yaml @@ -0,0 +1,5 @@ +include: _polish_few_shot_og_template_yaml +description: Poniżej znajdują się pytania wielokrotnego wyboru (z odpowiedziami) na + temat Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_polish_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8e1b953a1a05e6931e3a30b5d6e928284ac0823b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/include_base_44_polish_stem.yaml @@ -0,0 +1,5 @@ +include: _polish_few_shot_og_template_yaml +description: Poniżej znajdują się pytania wielokrotnego wyboru (z odpowiedziami) na + temat STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_polish_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Polish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_include_base_44_portuguese.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_include_base_44_portuguese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11bf7e84e264d9995c4718eca894a16d875f4711 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_include_base_44_portuguese.yaml @@ -0,0 +1,13 @@ +group: include_base_44_portuguese +task: +- include_base_44_portuguese_few_shot_og_stem +- include_base_44_portuguese_few_shot_og_social_science +- include_base_44_portuguese_few_shot_og_arts_humanities +- include_base_44_portuguese_few_shot_og_health_oriented_education +- include_base_44_portuguese_few_shot_og_business_commerce +- include_base_44_portuguese_few_shot_og_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_portuguese_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_portuguese_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ae882b22cd6f4a48c89737dfb7acc19b53add17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/_portuguese_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Portuguese +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab75c0347513a88f154b8414d0d7ab63d74038e7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_applied_science.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre Applied + Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_portuguese_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..946c4a2b76b594d4b2b7510a73a08fe7ccf2bf6f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre Arts + & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_portuguese_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f262c57e1fb8173c5da56c51b3e813c6cbe0d95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre Business + & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_portuguese_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ed5ce98ef0f484afad47bce4f5b98a4f6d277101 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre Health + oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_portuguese_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b960a247d5df613d9dd4c8706a0b4bb48b5fef08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_social_science.yaml @@ -0,0 +1,5 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre Social + Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_portuguese_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8cd4bcff92723b26b62d2ceae78be42ea65fa7d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/include_base_44_portuguese_stem.yaml @@ -0,0 +1,4 @@ +include: _portuguese_few_shot_og_template_yaml +description: A seguir estão perguntas de múltipla escolha (com respostas) sobre STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_portuguese_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Portuguese/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_include_base_44_russian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_include_base_44_russian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..461859a7e89a2dc1eae0cdbd87273ab4858a058f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_include_base_44_russian.yaml @@ -0,0 +1,15 @@ +group: include_base_44_russian +task: +- include_base_44_russian_few_shot_og_stem +- include_base_44_russian_few_shot_og_health_oriented_education +- include_base_44_russian_few_shot_og_arts_humanities +- include_base_44_russian_few_shot_og_driving_license +- include_base_44_russian_few_shot_og_social_science +- include_base_44_russian_few_shot_og_business_commerce +- include_base_44_russian_few_shot_og_marine_license +- include_base_44_russian_few_shot_og_applied_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_russian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_russian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b0e24871dfd767ebaa38747e9942f36c6222461 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/_russian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Russian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f0f0f2f8f39889dece8b9c0b20f9d2423c8ebd45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_applied_science.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Applied Science. +process_docs: !function 'utils.process_applied_science' +task: include_base_44_russian_few_shot_og_applied_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b4fd1ef353cf729a96095eb51d267f74d4476e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_russian_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_business_commerce.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_business_commerce.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6deb5b5974f26aa4f9cdb8f17f0243291d71a842 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_business_commerce.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Business & Commerce. +process_docs: !function 'utils.process_business_commerce' +task: include_base_44_russian_few_shot_og_business_commerce diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_driving_license.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_driving_license.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14b4b045a0df6f8abdb3cb7fdd1aef071b7b87e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_driving_license.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Driving License. +process_docs: !function 'utils.process_driving_license' +task: include_base_44_russian_few_shot_og_driving_license diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_health_oriented_education.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_health_oriented_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..09a6b2aacdca9a615a168042ff3a8ec55c705213 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_health_oriented_education.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Health oriented education. +process_docs: !function 'utils.process_health_oriented_education' +task: include_base_44_russian_few_shot_og_health_oriented_education diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..970b4699fa085f8e2e7cad8cba90382226498060 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/include_base_44_russian_social_science.yaml @@ -0,0 +1,5 @@ +include: _russian_few_shot_og_template_yaml +description: Ниже приведены вопросы с несколькими вариантами ответов (с ответами) + по теме Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_russian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Russian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_include_base_44_serbian.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_include_base_44_serbian.yaml new file mode 100644 index 0000000000000000000000000000000000000000..21a9c1b9ea982f79648a17d9c905cb32fafac962 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_include_base_44_serbian.yaml @@ -0,0 +1,10 @@ +group: include_base_44_serbian +task: +- include_base_44_serbian_few_shot_og_stem +- include_base_44_serbian_few_shot_og_arts_humanities +- include_base_44_serbian_few_shot_og_social_science +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_serbian_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_serbian_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b413a08ad6dd70cf46a1272743175fc1d0153fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/_serbian_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Serbian +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9c8f88d690f5a2e579d560fce2aa3e87d35a3d91 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_social_science.yaml @@ -0,0 +1,4 @@ +include: _serbian_few_shot_og_template_yaml +description: Следе питања са вишеструким избором (са одговорима) о Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_serbian_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4c4e22e88d0776320fec6f383f67a3127d87cd5d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/include_base_44_serbian_stem.yaml @@ -0,0 +1,4 @@ +include: _serbian_few_shot_og_template_yaml +description: Следе питања са вишеструким избором (са одговорима) о STEM. +process_docs: !function 'utils.process_stem' +task: include_base_44_serbian_few_shot_og_stem diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Serbian/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/include_base_44_spanish_social_science.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/include_base_44_spanish_social_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b46b0bf8f8fbbb55bb5e3ff421dff3d223e8a8a7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/include_base_44_spanish_social_science.yaml @@ -0,0 +1,5 @@ +include: _spanish_few_shot_og_template_yaml +description: Las siguientes son preguntas de opción múltiple (con respuestas) sobre + Social Science. +process_docs: !function 'utils.process_social_science' +task: include_base_44_spanish_few_shot_og_social_science diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Spanish/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_include_base_44_tagalog.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_include_base_44_tagalog.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ff41167d21e4f13d932b86c6687e73a230b9353c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_include_base_44_tagalog.yaml @@ -0,0 +1,9 @@ +group: include_base_44_tagalog +task: +- include_base_44_tagalog_few_shot_og_arts_humanities +- include_base_44_tagalog_few_shot_og_driving_license +aggregate_metric_list: +- metric: acc + weight_by_size: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_tagalog_few_shot_og_template_yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_tagalog_few_shot_og_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..88040e14d6f83717a48e22ed67c728bd2951f3c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/_tagalog_few_shot_og_template_yaml @@ -0,0 +1,18 @@ +dataset_path: CohereForAI/include-base-44 +dataset_name: Tagalog +test_split: test +output_type: multiple_choice +doc_to_text: "{{question.strip()}}\nA. {{option_a}}\nB. {{option_b}}\nC. {{option_c}}\n + D. {{option_d}}\nAnswer:" +doc_to_choice: + - A + - B + - C + - D +doc_to_target: answer +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 0.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/include_base_44_tagalog_arts_humanities.yaml b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/include_base_44_tagalog_arts_humanities.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f8bb4d417f6f1d61a6f01b75beb74eeed55e81f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tagalog/include_base_44_tagalog_arts_humanities.yaml @@ -0,0 +1,5 @@ +include: _tagalog_few_shot_og_template_yaml +description: Ang mga sumusunod ay maramihang pagpipiliang tanong (na may mga sagot) + tungkol sa Arts & Humanities. +process_docs: !function 'utils.process_arts_humanities' +task: include_base_44_tagalog_few_shot_og_arts_humanities diff --git a/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tamil/utils.py b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tamil/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..fde670c6bf543691c37830a7d87869023c70e447 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/include/few_shot_og/Tamil/utils.py @@ -0,0 +1,30 @@ +from functools import partial + + +CATEGORIES = [ + "Applied Science", + "Arts & Humanities", + "Business & Commerce", + "Driving License", + "General knowledge", + "Health oriented education", + "Marine License", + "Medical License", + "Professional certification", + "STEM", + "Social Science", +] + + +def process_docs(dataset, category): + return dataset.filter(lambda x: x["domain"] == category) + + +process_functions = { + f"process_{category.lower().replace(' & ', '_').replace(' ', '_')}": partial( + process_docs, category=category + ) + for category in CATEGORIES +} + +globals().update(process_functions) diff --git a/lm-evaluation-harness/lm_eval/tasks/japanese_leaderboard/ja_leaderboard_mgsm.yaml b/lm-evaluation-harness/lm_eval/tasks/japanese_leaderboard/ja_leaderboard_mgsm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bb041220a5ec95f66ef25aaab5fcceaef2cb6e58 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/japanese_leaderboard/ja_leaderboard_mgsm.yaml @@ -0,0 +1,32 @@ +task: ja_leaderboard_mgsm + +dataset_path: juletxara/mgsm +dataset_name: ja + +training_split: train +validation_split: null +test_split: test + +fewshot_split: train +num_fewshot: 5 + +description: "以下は、タスクを説明する指示と、文脈のある入力の組み合わせです。要求を適切に満たす応答を書きなさい。\n\n" +doc_to_text: "### 指示:\n与えられた問題に対して、ステップごとに答えを導き出してください。\n\n### 入力:\n{{ question | replace('問題:', '') }}\n\n### 応答:" +doc_to_target: "{{ answer | replace('ステップごとの答え:', '') }}" +target_delimiter: "\n" + +output_type: generate_until +process_results: !function ja_leaderboard_mgsm.process_results + +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + +generation_kwargs: + until: + - "\n\n" + do_sample: false + +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/jsonschema_bench/README.md b/lm-evaluation-harness/lm_eval/tasks/jsonschema_bench/README.md new file mode 100644 index 0000000000000000000000000000000000000000..30b67bdda484cdc24b099a68abbecc725b252f14 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/jsonschema_bench/README.md @@ -0,0 +1,97 @@ +# JSONSchema Bench + +## Tasks + +- `jsonschema_bench_easy`, corresponding to the `github_easy` split of the original paper +- `jsonschema_bench_medium`, corresponding to the `github_medium` split of the original paper +- `jsonschema_bench_hard`, corresponding to the `github_hard` split of the original paper + +Use `jsonschema_bench` tag to run all three tasks. + +## Metrics + +The JSONSchema Bench tasks are evaluated using the following two metrics: +- `json_validity`: This metric checks whether the generated output is valid JSON. It is a binary metric, where 1 indicates valid JSON and 0 indicates invalid JSON. We use `json` package to check the validity of the generated output. +- `schema_compliance`: This metric checks whether the generated output complies with the provided JSON schema. It is also a binary metric, where 1 indicates compliance and 0 indicates non-compliance. We use the `jsonschema` package to check the compliance of the generated output with the provided JSON schema. + +## Dependencies + +The JSONSchema Bench tasks require the `jsonschema` library to be installed. You can install it using pip: +```bash +pip install jsonschema\[format\] +``` + +The `format` extra is required to support the `format` keyword in JSON Schema, which is used in the tasks. + +## Sequence Length +The `easy` task requires a context window of 2K tokens, the `medium` task requires a context window of 3K tokens, and the `hard` task requires a context window of 10K tokens ( the exact number will vary depending on the tokenizer used, but 10K tokens is a good estimate). + +If you don't have enough memory to run the `hard` task, you can use the `--max_length` flag to reduce the context window size but this will truncate the schema and will lead to lower performance. + + +## Usage + +Here is an example of how to run 10 instances of the `jsonschema_bench_easy` task : +```bash +lm_eval \ + --model hf --gen_kwargs max_new_tokens=1024 \ + --model_args pretrained=meta-llama/Llama-3.2-1B-Instruct,parallelize=True\ + --tasks jsonschema_bench_medium \ + --batch_size auto \ + --limit 10 \ + --apply_chat_template \ + --fewshot_as_multiturn +``` + +The expected results is +``` +| Tasks |Version|Filter|n-shot| Metric | |Value| |Stderr| +|-----------------------|------:|------|-----:|-----------------|---|----:|---|-----:| +|jsonschema_bench_medium| 0.1|none | 2|json_validity |↑ | 1.0|± |0.0000| +| | |none | 2|schema_compliance|↑ | 0.2|± |0.1333| +``` + +## Dataset + +Available at [HF hub](https://huggingface.co/datasets/epfl-dlab/JSONSchemaBench) + +## Leaderboard + +We provide a [leaderboard](https://github.com/epfl-dlab/jsonschemabench-leaderboard) to track the progress of LLMs on the JSONSchema Bench tasks. +We welcome contributions to the leaderboard via pull requests. + + +## Paper + +JGenerating Structured Outputs from Language Models: Benchmark and Studies[https://arxiv.org/abs/2501.10868] + +Homepage: https://github.com/guidance-ai/jsonschemabench + + +## Citation +``` +@misc{geng2025jsonschemabench, + title={Generating Structured Outputs from Language Models: Benchmark and Studies}, + author={Saibo Geng and Hudson Cooper and Michał Moskal and Samuel Jenkins and Julian Berman and Nathan Ranchin and Robert West and Eric Horvitz and Harsha Nori}, + year={2025}, + eprint={2501.10868}, + archivePrefix={arXiv}, + primaryClass={cs.CL}, + url={https://arxiv.org/abs/2501.10868}, +} +``` + + + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [x] Is the task an existing benchmark in the literature? + * [x] Have you referenced the original paper that introduced the task? + * [x] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/README.md b/lm-evaluation-harness/lm_eval/tasks/kbl/README.md new file mode 100644 index 0000000000000000000000000000000000000000..c415a1d5db23d721493b643f567a423ca452409f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/README.md @@ -0,0 +1,127 @@ +# kbl + +### Paper + +Title: `Developing a Pragmatic Benchmark for Assessing Korean Legal Language Understanding in Large Language Models` + +Abstract: `Large language models (LLMs) have demonstrated remarkable performance in the legal domain, with GPT-4 even passing the Uniform Bar Exam in the U.S. However their efficacy remains limited for non-standardized tasks and tasks in languages other than English. This underscores the need for careful evaluation of LLMs within each legal system before application. Here, we introduce KBL, a benchmark for assessing the Korean legal language understanding of LLMs, consisting of (1) 7 legal knowledge tasks (510 examples), (2) 4 legal reasoning tasks (288 examples), and (3) the Korean bar exam (4 domains, 53 tasks, 2,510 examples). First two datasets were developed in close collaboration with lawyers to evaluate LLMs in practical scenarios in a certified manner. Furthermore, considering legal practitioners' frequent use of extensive legal documents for research, we assess LLMs in both a closed book setting, where they rely solely on internal knowledge, and a retrieval-augmented generation (RAG) setting, using a corpus of Korean statutes and precedents. The results indicate substantial room and opportunities for improvement.` + +`Korean Benchmark for Legal Language Understanding` + +Homepage: `https://github.com/lbox-kr/kbl` + + +### Citation + +``` +@inproceedings{kim2024kbl, + title = "Developing a Pragmatic Benchmark for Assessing {K}orean Legal Language Understanding in Large Language Models", + author = {Yeeun Kim and Young Rok Choi and Eunkyung Choi and Jinhwan Choi and Hai Jin Park and Wonseok Hwang}, + editor = "Al-Onaizan, Yaser and + Bansal, Mohit and + Chen, Yun-Nung", + booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2024", + month = nov, + year = "2024", + address = "Miami, Florida, USA", + publisher = "Association for Computational Linguistics", + url = "https://aclanthology.org/2024.findings-emnlp.319", + pages = "5573--5595", +} +``` + +### Groups, Tags, and Tasks + +#### Groups + +#### Tags + +* `kbl`: `All kbl tasks (7 knowledge, 4 reasoning, and 39 bar exam)` +* `kbl_knowledge_em`: `7 knowledge tasks` +* `kbl_reasoning_em`: `4 reasoning tasks` +* `kbl_bar_exam_em`: `53 bar exam tasks` +* `kbl_bar_exam_em_civil`: `13 bar exam tasks, civil law` +* `kbl_bar_exam_em_criminal`: `13 bar exam tasks, criminal law` +* `kbl_bar_exam_em_public`: `13 bar exam tasks, public law` +* `kbl_bar_exam_em_responsibility`: `14 bar exam tasks, professional responsibility (RESP) examination` + + +#### Tasks + +* `kbl_common_legal_mistake_qa_em`: `A QA task evaluating common legal misconceptions from the general public.` +* `kbl_knowledge_common_legal_mistake_qa_reasoning`: `Similar to 'kbl_common_legal_mistake_qa_em' but the answers are presented with correct/wrong rationals.` +* `kbl_knowledge_legal_concept_qa`: `A QA task addressing knowledge about complex legal concepts (legal terms).` +* `kbl_knowledge_offense_component_qa`: `A QA task evaluating whether a model knows specific actions meet the actual elements of a criminal offense.` +* `kbl_knowledge_query_and_statute_matching_qa`: `A QA task assessing whether the language model can accurately identify the relevant statute for a given query.` +* `kbl_knowledge_statute_hallucination_qa`: `A QA task evaluating whether a model can select the correct answer consists of a pair of (fictitious) statute and corresponding reasoning for given confusing legal questions.` +* `kbl_knowledge_statute_number_and_content_matching_qa`: `A QA dataset for evaluating where a model can accurately match the content of a law to its specific statute number.` +* `kbl_reasoning_case_relevance_qa_p`: `A QA task where a model needs to determine whether a given precedent is relavent to an input precedent.` +* `kbl_reasoning_case_relevance_qa_q`: `A QA task where a model needs to determine whether a given precedent is relavent to an input query.` +* `kbl_reasoning_causal_reasoning_qa`: `A QA task where a model needs to assess whether the defendant’s actions were the direct and decisive cause of the victim’s injury or death for each given factual description and claims.` +* `kbl_reasoning_statement_consistency_qa`: `A QA task where a model is required to accurately determine whether two presented statements are consistent with each other.` +* `bar_exam_civil_2012`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2013`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2014`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2015`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2016`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2017`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2018`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2019`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2020`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2021`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2022`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2023`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_civil_2024`: `Korean bar exam multiple-choice questions, civil law` +* `bar_exam_criminal_2012`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2013`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2014`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2015`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2016`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2017`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2018`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2019`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2020`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2021`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2022`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2023`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_criminal_2024`: `Korean bar exam multiple-choice questions, criminal law` +* `bar_exam_public_2012`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2013`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2014`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2015`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2016`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2017`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2018`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2019`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2020`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2021`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2022`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2023`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_public_2024`: `Korean bar exam multiple-choice questions, public law` +* `bar_exam_responsibility_2010`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2011`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2012`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2013`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2014`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2015`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2016`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2017`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2018`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2019`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2020`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2021`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2022`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` +* `bar_exam_responsibility_2023`: `Korean bar exam multiple-choice questions, professional responsibility (RESP) examination` + +### Checklist + +For adding novel benchmarks/datasets to the library: +* [ ] Is the task an existing benchmark in the literature? + * [ ] Have you referenced the original paper that introduced the task? + * [ ] If yes, does the original paper provide a reference implementation? If so, have you checked against the reference implementation and documented how to run such a test? + + +If other tasks on this dataset are already supported: +* [ ] Is the "Main" variant of this task clearly denoted? +* [ ] Have you provided a short sentence in a README on what each new variant adds / evaluates? +* [ ] Have you noted which, if any, published evaluation setups are matched by this variant? diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2013.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2013.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a7a8b537f65d279c3711d1d6c8ae9e5ea30280c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2013.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_civil_2013 +dataset_name: bar_exam_civil_2013 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2015.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2015.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3fe7ec896de1e62116241ba1496bdab62b9ce24 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2015.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_civil_2015 +dataset_name: bar_exam_civil_2015 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2016.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2016.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26f040c08f86bae57bc3a050a1978ed032d901ee --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2016.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_civil_2016 +dataset_name: bar_exam_civil_2016 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2020.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2020.yaml new file mode 100644 index 0000000000000000000000000000000000000000..89ff72747e800277502981607336ccd86e2fcc9c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2020.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_civil_2020 +dataset_name: bar_exam_civil_2020 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2023.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2023.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47342239c16d8e55571ed662bf44173bc1dfc4ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/civil/kbl_bar_exam_em_civil_2023.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_civil_2023 +dataset_name: bar_exam_civil_2023 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/_base_em_yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/_base_em_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b170dad754433986e81b180893f47354c3613aca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/_base_em_yaml @@ -0,0 +1,36 @@ +tag: + - kbl + - kbl_bar_exam_em + - kbl_bar_exam_em_criminal +description: '당신은 사용자의 질문에 친절하고 논리적으로 답변해 주는 법률 전문가 챗봇 입니다.\n' +dataset_path: lbox/kbl +test_split: test +output_type: generate_until +doc_to_text: '### 질문: {{question}} + + 다음 각 선택지를 읽고 A, B, C, D, E 중 하나를 선택하여 ''답변: A'' 와 같이 단답식으로 답해 주세요. + + A. {{A}} + + B. {{B}} + + C. {{C}} + + D. {{D}} + + E. {{E}} + + ### 답변:' +doc_to_target: gt +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: get-answer + filter: + - function: regex + regex_pattern: ([A-E]).* + - function: take_first diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2013.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2013.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6161688e620afbc3c385678c5701c3778b806ff0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2013.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2013 +dataset_name: bar_exam_criminal_2013 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2014.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2014.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dddb9ce258c7b7464ff01b8c094831b1da7461f4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2014.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2014 +dataset_name: bar_exam_criminal_2014 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2015.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2015.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db7607428f6c865d77f27b395e120d5746192f65 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2015.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2015 +dataset_name: bar_exam_criminal_2015 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2018.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2018.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bd1f6aa7f8c041d9c119a0f205c4091d1dc94b64 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2018.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2018 +dataset_name: bar_exam_criminal_2018 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2019.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2019.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e5de46ba6823575dc75100729363569f0c8d8d7a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2019.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2019 +dataset_name: bar_exam_criminal_2019 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2020.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2020.yaml new file mode 100644 index 0000000000000000000000000000000000000000..217c6783b9ecee68a1171092553014fe24b7f51f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2020.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2020 +dataset_name: bar_exam_criminal_2020 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2021.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2021.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e4527ba0e312635fd78985dd53ffcb39d3873425 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2021.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2021 +dataset_name: bar_exam_criminal_2021 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2023.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2023.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5b953dfb96e083c6318ed6882c8669bb4becda94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/criminal/kbl_bar_exam_em_criminal_2023.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_criminal_2023 +dataset_name: bar_exam_criminal_2023 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2012.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2012.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63c678ec3c1ff53dfe7f3fc3f6b4a96ddc7e0e15 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2012.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2012 +dataset_name: bar_exam_public_2012 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2013.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2013.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2af467a450f1b28b05e4bd52f8cb25a82035f4bd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2013.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2013 +dataset_name: bar_exam_public_2013 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2014.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2014.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0392f439a97795d7cf74cbc4318102c8f48b1ca5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2014.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2014 +dataset_name: bar_exam_public_2014 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2016.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2016.yaml new file mode 100644 index 0000000000000000000000000000000000000000..024b706e581bde80a55402be3e9b96cd13dab29b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2016.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2016 +dataset_name: bar_exam_public_2016 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2017.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2017.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50d172bab6914ccb849484eb39f602ae59faffba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2017.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2017 +dataset_name: bar_exam_public_2017 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2018.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2018.yaml new file mode 100644 index 0000000000000000000000000000000000000000..47341011d4589687093020ae118afb01d228e936 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2018.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2018 +dataset_name: bar_exam_public_2018 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2019.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2019.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d20fd4738328290c219435d4f314bb017f423b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2019.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2019 +dataset_name: bar_exam_public_2019 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2020.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2020.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5af0189c0eb32a21f03182c2ee9302baa8ceee72 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2020.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2020 +dataset_name: bar_exam_public_2020 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2021.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2021.yaml new file mode 100644 index 0000000000000000000000000000000000000000..02040f8431db39f7777a1a64884f7a83bb766712 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2021.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2021 +dataset_name: bar_exam_public_2021 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2022.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2022.yaml new file mode 100644 index 0000000000000000000000000000000000000000..00ec949c6e887dd294dab77eb3dfd47f872e9d8a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2022.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2022 +dataset_name: bar_exam_public_2022 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2023.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2023.yaml new file mode 100644 index 0000000000000000000000000000000000000000..27f8c4c718c6c6706f19833bd34fb42e00167cd2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2023.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2023 +dataset_name: bar_exam_public_2023 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2024.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2024.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdf0d9bf1894ab8f450f74259958a30588963017 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/public/kbl_bar_exam_em_public_2024.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_public_2024 +dataset_name: bar_exam_public_2024 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/_base_em_yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/_base_em_yaml new file mode 100644 index 0000000000000000000000000000000000000000..14350b0418bcf90ba031db8ca3b0ae858fc18663 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/_base_em_yaml @@ -0,0 +1,34 @@ +tag: + - kbl + - kbl_bar_exam_em + - kbl_bar_exam_em_responsibility +description: '당신은 사용자의 질문에 친절하고 논리적으로 답변해 주는 법률 전문가 챗봇 입니다.\n' +dataset_path: lbox/kbl +test_split: test +output_type: generate_until +doc_to_text: '### 질문: {{question}} + + 다음 각 선택지를 읽고 A, B, C, D 중 하나를 선택하여 ''답변: A'' 와 같이 단답식으로 답해 주세요. + + A. {{A}} + + B. {{B}} + + C. {{C}} + + D. {{D}} + + ### 답변:' +doc_to_target: gt +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: get-answer + filter: + - function: regex + regex_pattern: ([A-D]).* + - function: take_first diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2010.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2010.yaml new file mode 100644 index 0000000000000000000000000000000000000000..11efbd5cddec8f1b3b8cee5cd6fe1d2e7cefb979 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2010.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_responsibility_2010 +dataset_name: bar_exam_responsibility_2010 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2011.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2011.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd75d0039e547134e4709d3c2d5c8787ec6c9969 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2011.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_responsibility_2011 +dataset_name: bar_exam_responsibility_2011 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2013.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2013.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0d276113502555d34c896ecaaf592bc874475b47 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2013.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_responsibility_2013 +dataset_name: bar_exam_responsibility_2013 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2014.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2014.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c14782c90dd42ccc033dcbab298f274378ad5469 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2014.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_responsibility_2014 +dataset_name: bar_exam_responsibility_2014 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2022.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2022.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e23b221f46bb80394326160cf25f4692581331f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/bar_exam/responsibility/kbl_bar_exam_em_responsibility_2022.yaml @@ -0,0 +1,3 @@ +task: kbl_bar_exam_em_responsibility_2022 +dataset_name: bar_exam_responsibility_2022 +include: _base_em_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kbl/knowledge/kbl_statute_hallucination_qa_em.yaml b/lm-evaluation-harness/lm_eval/tasks/kbl/knowledge/kbl_statute_hallucination_qa_em.yaml new file mode 100644 index 0000000000000000000000000000000000000000..efa5082f7012782cb6b9fe735e60310e14a69f48 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kbl/knowledge/kbl_statute_hallucination_qa_em.yaml @@ -0,0 +1,4 @@ +task: kbl_statute_hallucination_qa_em +dataset_name: kbl_knowledge_statute_hallucination_qa +doc_to_text: "### 질문: {{question}}\nA. {{A}}\nB. {{B}}\nC. {{C}}\nD. {{D}}\n'A', 'B', 'C', 'D' 중 하나를 선택하여 ''답변: A'' 와 같이 단답식으로 답해 주세요." +include: _kbl_knowledge_yaml diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_civil_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_civil_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8c5064b93cabcd54fd7f1d7fa5fb1380712a11d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_civil_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: Civil-Engineering +include: _direct_kmmlu_yaml +task: kmmlu_direct_civil_engineering +tag: kmmlu_direct_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_environmental_science.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_environmental_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..104a4b9ed9045bb9674201434626f565b1ce3a1c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_environmental_science.yaml @@ -0,0 +1,4 @@ +dataset_name: Environmental-Science +include: _direct_kmmlu_yaml +task: kmmlu_direct_environmental_science +tag: kmmlu_direct_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_information_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_information_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50fc6e91f00b6b15c6861f71fad50be41f79c251 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_information_technology.yaml @@ -0,0 +1,4 @@ +dataset_name: Information-Technology +include: _direct_kmmlu_yaml +task: kmmlu_direct_information_technology +tag: kmmlu_direct_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_interior_architecture_and_design.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_interior_architecture_and_design.yaml new file mode 100644 index 0000000000000000000000000000000000000000..638de434507ea7dd7b93fd78a1636cf09ae4fae1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_interior_architecture_and_design.yaml @@ -0,0 +1,4 @@ +dataset_name: Interior-Architecture-and-Design +include: _direct_kmmlu_yaml +task: kmmlu_direct_interior_architecture_and_design +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_machine_design_and_manufacturing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_machine_design_and_manufacturing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..587d25d0e4fce1abccbc5ee12e4613f77609d664 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_machine_design_and_manufacturing.yaml @@ -0,0 +1,4 @@ +dataset_name: Machine-Design-and-Manufacturing +include: _direct_kmmlu_yaml +task: kmmlu_direct_machine_design_and_manufacturing +tag: kmmlu_direct_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_management.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..aec441bb022279edca8157e2347507173e37ca02 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_management.yaml @@ -0,0 +1,4 @@ +dataset_name: Management +include: _direct_kmmlu_yaml +task: kmmlu_direct_management +tag: kmmlu_direct_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..10dadc008401647e1e9c58e034937abb5b918f4f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_marketing.yaml @@ -0,0 +1,4 @@ +dataset_name: Marketing +include: _direct_kmmlu_yaml +task: kmmlu_direct_marketing +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_nondestructive_testing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_nondestructive_testing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e37bd1c1ca6dbb97b4ee35a6191304e39c97bcd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_nondestructive_testing.yaml @@ -0,0 +1,4 @@ +dataset_name: Nondestructive-Testing +include: _direct_kmmlu_yaml +task: kmmlu_direct_nondestructive_testing +tag: kmmlu_direct_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_patent.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_patent.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e829b99583a0f125856b2385b0bce6c5130c775d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_patent.yaml @@ -0,0 +1,4 @@ +dataset_name: Patent +include: _direct_kmmlu_yaml +task: kmmlu_direct_patent +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_political_science_and_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_political_science_and_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..adf6c1b7f2b1bebfb57cd27378cd08475fc4fa2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_political_science_and_sociology.yaml @@ -0,0 +1,4 @@ +dataset_name: Political-Science-and-Sociology +include: _direct_kmmlu_yaml +task: kmmlu_direct_political_science_and_sociology +tag: kmmlu_direct_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_public_safety.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_public_safety.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5926a45c96b701637a1a6d712449268ca1f118dc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_public_safety.yaml @@ -0,0 +1,4 @@ +dataset_name: Public-Safety +include: _direct_kmmlu_yaml +task: kmmlu_direct_public_safety +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_real_estate.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_real_estate.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e8872a53035ba58ce8ed6a94d19c0c5d65f5d96c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_real_estate.yaml @@ -0,0 +1,4 @@ +dataset_name: Real-Estate +include: _direct_kmmlu_yaml +task: kmmlu_direct_real_estate +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_refrigerating_machinery.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_refrigerating_machinery.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7378739041e20ef91e64eb4c7a9f0fb42032fc21 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_refrigerating_machinery.yaml @@ -0,0 +1,4 @@ +dataset_name: Refrigerating-Machinery +include: _direct_kmmlu_yaml +task: kmmlu_direct_refrigerating_machinery +tag: kmmlu_direct_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_social_welfare.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_social_welfare.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52f731fb370863cabe5895be26ea128790cab0b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_social_welfare.yaml @@ -0,0 +1,4 @@ +dataset_name: Social-Welfare +include: _direct_kmmlu_yaml +task: kmmlu_direct_social_welfare +tag: kmmlu_direct_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_taxation.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_taxation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..caa0d7984173349a2e4e7f5cd1f4a8b86a107726 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_taxation.yaml @@ -0,0 +1,4 @@ +dataset_name: Taxation +include: _direct_kmmlu_yaml +task: kmmlu_direct_taxation +tag: kmmlu_direct_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_telecommunications_and_wireless_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_telecommunications_and_wireless_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f98b1d4984b352c7f872292a7ac4e2cd9a7fdae --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct/kmmlu_direct_telecommunications_and_wireless_technology.yaml @@ -0,0 +1,4 @@ +dataset_name: Telecommunications-and-Wireless-Technology +include: _direct_kmmlu_yaml +task: kmmlu_direct_telecommunications_and_wireless_technology +tag: kmmlu_direct_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_direct_hard_kmmlu_yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_direct_hard_kmmlu_yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5ed0fda26293003d9ccd37c54d0f4d76da7eea2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_direct_hard_kmmlu_yaml @@ -0,0 +1,24 @@ +dataset_path: HAERAE-HUB/KMMLU-HARD +output_type: generate_until +test_split: test +fewshot_split: dev +doc_to_text: "{{question.strip()}}\nA. {{A}}\nB. {{B}}\nC. {{C}}\nD. {{D}}\n정답:" +doc_to_target: "{{['A', 'B', 'C', 'D'][answer-1]}}" +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - " " +generation_kwargs: + until: + - "Q:" + - "\n\n" + - "" + - "." + do_sample: false + temperature: 0.0 +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54206cdb779c2d7354f9d676731bf8d544a10ab6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard.yaml @@ -0,0 +1,11 @@ +group: kmmlu_direct_hard +task: + - kmmlu_direct_hard_stem + - kmmlu_direct_hard_other + - kmmlu_direct_hard_applied_science + - kmmlu_direct_hard_humss +aggregate_metric_list: + - metric: exact_match + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0f70ae139dd537613f70dd069cacb535a231e44a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_applied_science.yaml @@ -0,0 +1,8 @@ +group: kmmlu_direct_hard_applied_science +task: + - kmmlu_direct_hard_applied_science_tasks +aggregate_metric_list: + - metric: exact_match + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_humss.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_humss.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b28fdd1522ba82a26d71af523682fbc93d0a6656 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_humss.yaml @@ -0,0 +1,8 @@ +group: kmmlu_direct_hard_humss +task: + - kmmlu_direct_hard_humss_tasks +aggregate_metric_list: + - metric: exact_match + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_other.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f216caa648596d1c1ff3bf2597c04352ae1292c5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/_kmmlu_direct_hard_other.yaml @@ -0,0 +1,8 @@ +group: kmmlu_direct_hard_other +task: + - kmmlu_direct_hard_other_tasks +aggregate_metric_list: + - metric: exact_match + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d92b933d4bf31038e2aba5339a7ed5de95acf82c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_accounting.yaml @@ -0,0 +1,4 @@ +dataset_name: accounting +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_accounting +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_agricultural_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_agricultural_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d78427d0211db3d8c5a7fbb1c4a93a612416c86d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_agricultural_sciences.yaml @@ -0,0 +1,4 @@ +dataset_name: agricultural_sciences +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_agricultural_sciences +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_aviation_engineering_and_maintenance.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_aviation_engineering_and_maintenance.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6713f04da2495d8790d768e79e13b33ef057433a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_aviation_engineering_and_maintenance.yaml @@ -0,0 +1,4 @@ +dataset_name: aviation_engineering_and_maintenance +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_aviation_engineering_and_maintenance +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e98a380f9255dd6afe739a0817361907125e54c4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_biology.yaml @@ -0,0 +1,4 @@ +dataset_name: biology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_biology +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b505e3175f3e9459c8694f9216497c4415e0abf3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemical_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: chemical_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_chemical_engineering +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d805e2340f321dc4a1b0a1b1fca7e1eed5c2e77a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_chemistry.yaml @@ -0,0 +1,4 @@ +dataset_name: chemistry +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_chemistry +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_civil_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_civil_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30622d50c6811bcfc4bba3a35aa0f6d29246ad0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_civil_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: civil_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_civil_engineering +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_construction.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_construction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e050e106754accd43523f8dcb1facee64b4ac27b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_construction.yaml @@ -0,0 +1,4 @@ +dataset_name: construction +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_construction +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_criminal_law.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_criminal_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3072b6f0b538a8fc33e2e410da3a43669626c444 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_criminal_law.yaml @@ -0,0 +1,4 @@ +dataset_name: criminal_law +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_criminal_law +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_ecology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_ecology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3129f467d25af5d389380cefe92b2345f8bf78ff --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_ecology.yaml @@ -0,0 +1,4 @@ +dataset_name: ecology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_ecology +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..87069840e66e636cd2627ed3fb574b1b91019892 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_economics.yaml @@ -0,0 +1,4 @@ +dataset_name: economics +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_economics +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_education.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..75baa1364b434443285205073e1192b195670e17 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_education.yaml @@ -0,0 +1,4 @@ +dataset_name: education +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_education +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electrical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electrical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..789cdfb81cd5f34842fc2985fb8927e82162a8e4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electrical_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: electrical_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_electrical_engineering +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electronics_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electronics_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a1736e0b2584d208cfb90ed777d8de10030e145 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_electronics_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: electronics_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_electronics_engineering +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_environmental_science.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_environmental_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60c0253e0f13c9608f22ad5a58a3e10f8527053c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_environmental_science.yaml @@ -0,0 +1,4 @@ +dataset_name: environmental_science +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_environmental_science +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_fashion.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_fashion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86bbb9b49c03165daff8c6588f9ca053e798c678 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_fashion.yaml @@ -0,0 +1,4 @@ +dataset_name: fashion +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_fashion +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_food_processing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_food_processing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6b2817d2c0f38faa9c9d46e0f40385a1f928293c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_food_processing.yaml @@ -0,0 +1,4 @@ +dataset_name: food_processing +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_food_processing +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_gas_technology_and_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_gas_technology_and_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c2d2f4772b86a764f4c73ef391a31acd7a68f787 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_gas_technology_and_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: gas_technology_and_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_gas_technology_and_engineering +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_geomatics.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_geomatics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9dadc72dc31bf2b7ced96c940a2bdcf3d8ab681f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_geomatics.yaml @@ -0,0 +1,4 @@ +dataset_name: geomatics +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_geomatics +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_health.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_health.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f1bf4c778c8c5ef24d135833edfae70e09cc138d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_health.yaml @@ -0,0 +1,4 @@ +dataset_name: health +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_health +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_industrial_engineer.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_industrial_engineer.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f7b73ea5fa64072648fcedf88406a66b695ca74 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_industrial_engineer.yaml @@ -0,0 +1,4 @@ +dataset_name: industrial_engineer +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_industrial_engineer +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_information_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_information_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a1c5cf9dbf3369475f573419f127cd386153017e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_information_technology.yaml @@ -0,0 +1,4 @@ +dataset_name: information_technology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_information_technology +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_interior_architecture_and_design.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_interior_architecture_and_design.yaml new file mode 100644 index 0000000000000000000000000000000000000000..65a20727fc67f1a28db33968384e343ea12d3fc1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_interior_architecture_and_design.yaml @@ -0,0 +1,4 @@ +dataset_name: interior_architecture_and_design +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_interior_architecture_and_design +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_korean_history.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_korean_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c10a9f576ff6e633a9fa7fabb7eebdff1fb01728 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_korean_history.yaml @@ -0,0 +1,4 @@ +dataset_name: korean_history +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_korean_history +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_law.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..96e5514f25195742e62c7632625d0e9e0506a2fe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_law.yaml @@ -0,0 +1,4 @@ +dataset_name: law +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_law +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_machine_design_and_manufacturing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_machine_design_and_manufacturing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50dfd63b230bb0fb68abe387f856be5121f0a5c3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_machine_design_and_manufacturing.yaml @@ -0,0 +1,4 @@ +dataset_name: machine_design_and_manufacturing +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_machine_design_and_manufacturing +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_management.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48c339d7439af8e827aef5ec0e0a8391209f86bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_management.yaml @@ -0,0 +1,4 @@ +dataset_name: management +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_management +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_maritime_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_maritime_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..937bfd27f20d790f4c768503917c2455f891244e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_maritime_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: maritime_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_maritime_engineering +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1ae4088a1687349fcfc65636610636ecfc96b2f1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_marketing.yaml @@ -0,0 +1,4 @@ +dataset_name: marketing +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_marketing +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_materials_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_materials_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..432460ebf7b0368ad5b51f69bce0fd80c6f582fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_materials_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: materials_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_materials_engineering +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_math.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_math.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53d2fca14d3d29b1423d2f54b30831ba98dd9d33 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_math.yaml @@ -0,0 +1,4 @@ +dataset_name: math +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_math +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_mechanical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_mechanical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a3994ea59183b14325aa6741a39096517cbcb4e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_mechanical_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: mechanical_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_mechanical_engineering +tag: kmmlu_direct_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_nondestructive_testing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_nondestructive_testing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..909c502c02556e3a57410355b6e433ff24a03f0d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_nondestructive_testing.yaml @@ -0,0 +1,4 @@ +dataset_name: nondestructive_testing +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_nondestructive_testing +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_patent.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_patent.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d8faf9723755ef53d8be2bdac483009abd10cf12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_patent.yaml @@ -0,0 +1,4 @@ +dataset_name: patent +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_patent +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_political_science_and_sociology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_political_science_and_sociology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0b6505074663bc8e98b391462b8f4b41b924c4cd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_political_science_and_sociology.yaml @@ -0,0 +1,4 @@ +dataset_name: political_science_and_sociology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_political_science_and_sociology +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_psychology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_psychology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b1a6f7777f22d01310c4b798f44a72b7aa3c7f9b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_psychology.yaml @@ -0,0 +1,4 @@ +dataset_name: psychology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_psychology +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_public_safety.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_public_safety.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3da462946a87b77ab21887c8ca2cf1e1ba26bfb8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_public_safety.yaml @@ -0,0 +1,4 @@ +dataset_name: public_safety +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_public_safety +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_railway_and_automotive_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_railway_and_automotive_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74e5e02f43da2789a7e99481917160ddf5f369ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_railway_and_automotive_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: railway_and_automotive_engineering +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_railway_and_automotive_engineering +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_real_estate.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_real_estate.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f23fae524939eba477f18d87347c489dc20183f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_real_estate.yaml @@ -0,0 +1,4 @@ +dataset_name: real_estate +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_real_estate +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_refrigerating_machinery.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_refrigerating_machinery.yaml new file mode 100644 index 0000000000000000000000000000000000000000..192a1f2c0da7c395e2f6f99b08e3e6bcdf822f94 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_refrigerating_machinery.yaml @@ -0,0 +1,4 @@ +dataset_name: refrigerating_machinery +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_refrigerating_machinery +tag: kmmlu_direct_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_social_welfare.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_social_welfare.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c24babc33af228426aeab675ad0c60fefdd90255 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_social_welfare.yaml @@ -0,0 +1,4 @@ +dataset_name: social_welfare +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_social_welfare +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_taxation.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_taxation.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17586af6d69cb01eeeec768b2d22ee8a0755b316 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_taxation.yaml @@ -0,0 +1,4 @@ +dataset_name: taxation +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_taxation +tag: kmmlu_direct_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_telecommunications_and_wireless_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_telecommunications_and_wireless_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bed0df91c97f7fab0f76210137687c765a834f4d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/direct_hard/kmmlu_direct_hard_telecommunications_and_wireless_technology.yaml @@ -0,0 +1,4 @@ +dataset_name: telecommunications_and_wireless_technology +include: _direct_hard_kmmlu_yaml +task: kmmlu_direct_hard_telecommunications_and_wireless_technology +tag: kmmlu_direct_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_hard_kmmlu_yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_hard_kmmlu_yaml new file mode 100644 index 0000000000000000000000000000000000000000..b3e6970527d9903d7bcf29397d30f7237a60679c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_hard_kmmlu_yaml @@ -0,0 +1,13 @@ +dataset_path: HAERAE-HUB/KMMLU-HARD +output_type: multiple_choice +test_split: test +fewshot_split: dev +doc_to_text: "{{question.strip()}}\nA. {{A}}\nB. {{B}}\nC. {{C}}\nD. {{D}}\n정답:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: "{{answer-1}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard.yaml new file mode 100644 index 0000000000000000000000000000000000000000..827e74ec100edf3ce40d55e510a93a248ad48926 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard.yaml @@ -0,0 +1,11 @@ +group: kmmlu_hard +task: + - kmmlu_hard_stem + - kmmlu_hard_other + - kmmlu_hard_applied_science + - kmmlu_hard_humss +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_applied_science.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_applied_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..76d383af040a29ac9d0944c431dc643179d93493 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_applied_science.yaml @@ -0,0 +1,8 @@ +group: kmmlu_hard_applied_science +task: + - kmmlu_hard_applied_science_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_humss.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_humss.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39eb5a7a2621b40ede548d1bf31ccdb42917c333 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_humss.yaml @@ -0,0 +1,8 @@ +group: kmmlu_hard_humss +task: + - kmmlu_hard_humss_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_other.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_other.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5759fe8844654211f1535fd570dd64d8d607f870 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_other.yaml @@ -0,0 +1,8 @@ +group: kmmlu_hard_other +task: + - kmmlu_hard_other_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_stem.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_stem.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee14c726413524214d9719cee95a50aa9d1cd621 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/_kmmlu_hard_stem.yaml @@ -0,0 +1,8 @@ +group: kmmlu_hard_stem +task: + - kmmlu_hard_stem_tasks +aggregate_metric_list: + - metric: acc + weight_by_size: True +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_accounting.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_accounting.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0c341baac0f7897998f48f1e6b3553023e18ef95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_accounting.yaml @@ -0,0 +1,4 @@ +dataset_name: accounting +include: _hard_kmmlu_yaml +task: kmmlu_hard_accounting +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_agricultural_sciences.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_agricultural_sciences.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90d284c8f700f0256454a9a73c19b32a64e41b38 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_agricultural_sciences.yaml @@ -0,0 +1,4 @@ +dataset_name: agricultural_sciences +include: _hard_kmmlu_yaml +task: kmmlu_hard_agricultural_sciences +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_aviation_engineering_and_maintenance.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_aviation_engineering_and_maintenance.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5ec90f362f971a7e7c08d304eabd53f0b0762759 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_aviation_engineering_and_maintenance.yaml @@ -0,0 +1,4 @@ +dataset_name: aviation_engineering_and_maintenance +include: _hard_kmmlu_yaml +task: kmmlu_hard_aviation_engineering_and_maintenance +tag: kmmlu_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_biology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_biology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..045e17e7807bd982ffac12bb2375b7126bdef024 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_biology.yaml @@ -0,0 +1,4 @@ +dataset_name: biology +include: _hard_kmmlu_yaml +task: kmmlu_hard_biology +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemical_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemical_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cbfa42eb2041e55008503afd0034ad096fc975f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemical_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: chemical_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_chemical_engineering +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemistry.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemistry.yaml new file mode 100644 index 0000000000000000000000000000000000000000..67c65d659834015cfdd0315bb9244debb1aacf45 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_chemistry.yaml @@ -0,0 +1,4 @@ +dataset_name: chemistry +include: _hard_kmmlu_yaml +task: kmmlu_hard_chemistry +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_civil_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_civil_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..58e3c87a84b5ea7e8e4a3e5278c12b959ace12ef --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_civil_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: civil_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_civil_engineering +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_computer_science.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_computer_science.yaml new file mode 100644 index 0000000000000000000000000000000000000000..42f91467679c3c46fa7c05ca296c44974c05feec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_computer_science.yaml @@ -0,0 +1,4 @@ +dataset_name: computer_science +include: _hard_kmmlu_yaml +task: kmmlu_hard_computer_science +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_construction.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_construction.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55a5a1d0d99d889eaff35f030b9800a58b332f56 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_construction.yaml @@ -0,0 +1,4 @@ +dataset_name: construction +include: _hard_kmmlu_yaml +task: kmmlu_hard_construction +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_criminal_law.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_criminal_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..14e4d5ad65b838d3df1456c142f5588fa4b43806 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_criminal_law.yaml @@ -0,0 +1,4 @@ +dataset_name: criminal_law +include: _hard_kmmlu_yaml +task: kmmlu_hard_criminal_law +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_ecology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_ecology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c737b1abaf7c55898a1123d4d4161aab10651cd9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_ecology.yaml @@ -0,0 +1,4 @@ +dataset_name: ecology +include: _hard_kmmlu_yaml +task: kmmlu_hard_ecology +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_economics.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_economics.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9a0084dc3867f40e85a1b82d6a2d9e0df6725b99 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_economics.yaml @@ -0,0 +1,4 @@ +dataset_name: economics +include: _hard_kmmlu_yaml +task: kmmlu_hard_economics +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_education.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_education.yaml new file mode 100644 index 0000000000000000000000000000000000000000..568d094d67e99094758faebc31e3ef441b1d73f6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_education.yaml @@ -0,0 +1,4 @@ +dataset_name: education +include: _hard_kmmlu_yaml +task: kmmlu_hard_education +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_electronics_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_electronics_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..843c92a056a5d4a47e610f1c1d5cdc144ff4305b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_electronics_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: electronics_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_electronics_engineering +tag: kmmlu_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_fashion.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_fashion.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3ba973ba6a03648f05302b424ad809ff2a1571bb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_fashion.yaml @@ -0,0 +1,4 @@ +dataset_name: fashion +include: _hard_kmmlu_yaml +task: kmmlu_hard_fashion +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_food_processing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_food_processing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd08fe3b99d93cf6aa2dfde0deed6072a6b79478 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_food_processing.yaml @@ -0,0 +1,4 @@ +dataset_name: food_processing +include: _hard_kmmlu_yaml +task: kmmlu_hard_food_processing +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_gas_technology_and_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_gas_technology_and_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fe30680ae6fc63d75c5ab870cb81b38eab6b21b8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_gas_technology_and_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: gas_technology_and_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_gas_technology_and_engineering +tag: kmmlu_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_health.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_health.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dcd2b179d6b50e5ddc0c642289350535bf86089f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_health.yaml @@ -0,0 +1,4 @@ +dataset_name: health +include: _hard_kmmlu_yaml +task: kmmlu_hard_health +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_industrial_engineer.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_industrial_engineer.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2e8449ffd7128f4c98f91605f9d53647363ece9d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_industrial_engineer.yaml @@ -0,0 +1,4 @@ +dataset_name: industrial_engineer +include: _hard_kmmlu_yaml +task: kmmlu_hard_industrial_engineer +tag: kmmlu_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_information_technology.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_information_technology.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86ded35de128bf2e34ef831bc85e7aac4beb4373 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_information_technology.yaml @@ -0,0 +1,4 @@ +dataset_name: information_technology +include: _hard_kmmlu_yaml +task: kmmlu_hard_information_technology +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_korean_history.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_korean_history.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4d4152b7945d76ad61e2033c3001d3b326f898dd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_korean_history.yaml @@ -0,0 +1,4 @@ +dataset_name: korean_history +include: _hard_kmmlu_yaml +task: kmmlu_hard_korean_history +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_law.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_law.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a75d9041c850f9750e588eea9683efe9e682497 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_law.yaml @@ -0,0 +1,4 @@ +dataset_name: law +include: _hard_kmmlu_yaml +task: kmmlu_hard_law +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_management.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_management.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3f27519e2d3511152bc9ac778b8c7aa615b9cad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_management.yaml @@ -0,0 +1,4 @@ +dataset_name: management +include: _hard_kmmlu_yaml +task: kmmlu_hard_management +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_maritime_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_maritime_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dec43bc8045ca1ac93c38e53b6b43904051a3722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_maritime_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: maritime_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_maritime_engineering +tag: kmmlu_hard_applied_science_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_marketing.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_marketing.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f86cfe17bc530b2e54c658b70da9d8f8499cc7d5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_marketing.yaml @@ -0,0 +1,4 @@ +dataset_name: marketing +include: _hard_kmmlu_yaml +task: kmmlu_hard_marketing +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_materials_engineering.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_materials_engineering.yaml new file mode 100644 index 0000000000000000000000000000000000000000..684120a077fa5616eb52cbdee9c1603b60ab135b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_materials_engineering.yaml @@ -0,0 +1,4 @@ +dataset_name: materials_engineering +include: _hard_kmmlu_yaml +task: kmmlu_hard_materials_engineering +tag: kmmlu_hard_stem_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_patent.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_patent.yaml new file mode 100644 index 0000000000000000000000000000000000000000..910f11c54c781cc0f443e83827e09b2d0790775d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_patent.yaml @@ -0,0 +1,4 @@ +dataset_name: patent +include: _hard_kmmlu_yaml +task: kmmlu_hard_patent +tag: kmmlu_hard_other_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_social_welfare.yaml b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_social_welfare.yaml new file mode 100644 index 0000000000000000000000000000000000000000..24f105e4677e7461087037e51f9f66add272fb35 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kmmlu/hard/kmmlu_hard_social_welfare.yaml @@ -0,0 +1,4 @@ +dataset_name: social_welfare +include: _hard_kmmlu_yaml +task: kmmlu_hard_social_welfare +tag: kmmlu_hard_humss_tasks diff --git a/lm-evaluation-harness/lm_eval/tasks/kobest/README.md b/lm-evaluation-harness/lm_eval/tasks/kobest/README.md new file mode 100644 index 0000000000000000000000000000000000000000..5a160da77140f37244dde849f42ab5b3f223a0a4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kobest/README.md @@ -0,0 +1,37 @@ +# LAMBADA + +### Paper +Title: `KOBEST: Korean Balanced Evaluation of Significant Tasks` + +Abstract: https://arxiv.org/abs/2204.04541 + +A well-formulated benchmark plays a critical role in spurring advancements in the natural language processing (NLP) field, as it allows objective and precise evaluation of diverse models. As modern language models (LMs) have become more elaborate and sophisticated, more difficult benchmarks that require linguistic knowledge and reasoning have been proposed. However, most of these benchmarks only support English, and great effort is necessary to construct benchmarks for other low resource languages. To this end, we propose a new benchmark named Korean balanced evaluation of significant tasks (KoBEST), which consists of five Korean-language downstream tasks. Professional Korean linguists designed the tasks that require advanced Korean linguistic knowledge. Moreover, our data is purely annotated by humans and thoroughly reviewed to guarantee high data quality. We also provide baseline models and human performance results. Our dataset is available on the Huggingface. + + +Homepage: https://huggingface.co/datasets/skt/kobest_v1 + +### Groups and Tasks + +#### Groups + +- `kobest` + +#### Tasks + +- `kobest_boolq` +- `kobest_copa` +- `kobest_hallawag` +- `kobest_sentineg` +- `kobest_wic` + + +### Citation + +@misc{ + author={Dohyeong Kim, Myeongjun Jang, Deuk Sin Kwon, Eric Davis}, + title={KOBEST: Korean Balanced Evaluation of Significant Tasks}, + DOI={https://doi.org/10.48550/arXiv.2204.04541}, + publisher={arXiv}, + year={2022}, + month={Apr} +} diff --git a/lm-evaluation-harness/lm_eval/tasks/kobest/_kobest.yaml b/lm-evaluation-harness/lm_eval/tasks/kobest/_kobest.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf23f6643a37a0031c947c67474b0197eb864f28 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/kobest/_kobest.yaml @@ -0,0 +1,19 @@ +group: kobest +task: + - kobest_boolq + - kobest_copa + - kobest_hellaswag + - kobest_sentineg + - kobest_wic +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true + - metric: f1 + aggregation: mean + weight_by_size: true +metadata: + version: 1.0