diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/hf_audiolm.cpython-311.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/hf_audiolm.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ff0c6683692c6c3239f8204a2c9dc88139bd0ba0 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/hf_audiolm.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/huggingface.cpython-311.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/huggingface.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..15e73b486a0e77687ada2451257a120cc2a7ba39 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/huggingface.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/nemo_lm.cpython-310.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/nemo_lm.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..223c04c2c1dc6608e7e77dc65878e538a9c1a514 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/nemo_lm.cpython-310.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/neuron_optimum.cpython-310.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/neuron_optimum.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..be5b79ecc0abec36da04110c5f0636b852cab377 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/neuron_optimum.cpython-310.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-310.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ffff6716d270dfdcf723d2fc62a299fc9f0b8c77 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-310.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-311.pyc b/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..569ece42e5994565dde42255e2e68cc7be09d869 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/models/__pycache__/optimum_lm.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/models/huggingface.py b/lm-evaluation-harness/lm_eval/models/huggingface.py new file mode 100644 index 0000000000000000000000000000000000000000..a6231570ad9021237ac04b3a581a00d11b8bf69d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/models/huggingface.py @@ -0,0 +1,1480 @@ +import copy +import logging +import os +from datetime import timedelta +from pathlib import Path +from typing import Any, Dict, List, Literal, Optional, Tuple, Union + +import jinja2 +import torch +import torch.nn.functional as F +import transformers +from accelerate import ( + Accelerator, + InitProcessGroupKwargs, + find_executable_batch_size, +) +from accelerate.utils import get_max_memory +from huggingface_hub import HfApi +from packaging import version +from peft import PeftModel +from peft import __version__ as PEFT_VERSION +from tqdm import tqdm +from transformers.models.auto.modeling_auto import ( + MODEL_FOR_CAUSAL_LM_MAPPING_NAMES, + MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING_NAMES, +) + +from lm_eval import utils +from lm_eval.api.instance import Instance +from lm_eval.api.model import TemplateLM +from lm_eval.api.registry import register_model +from lm_eval.models.utils import ( + Collator, + clear_torch_cache, + configure_pad_token, + get_dtype, + handle_stop_sequences, + pad_and_concat, + stop_sequences_criteria, +) + + +eval_logger = logging.getLogger(__name__) + + +@register_model("hf-auto", "hf", "huggingface") +class HFLM(TemplateLM): + """ + An abstracted Huggingface model class. Enables usage with both models of + `transformers.AutoModelForCausalLM` and `transformers.AutoModelForSeq2SeqLM` classes. + + Supports data-parallel multi-GPU with HF Accelerate. + """ + + AUTO_MODEL_CLASS = None + _DEFAULT_MAX_LENGTH = 2048 + + def __init__( + self, + pretrained: Union[str, transformers.PreTrainedModel], + backend: Literal["default", "causal", "seq2seq"] = "default", + # override whether the model should be treated as decoder-only (causal) or encoder-decoder (seq2seq) + revision: Optional[str] = "main", + subfolder: Optional[str] = None, + tokenizer: Optional[ + Union[ + str, + transformers.PreTrainedTokenizer, + transformers.PreTrainedTokenizerFast, + ] + ] = None, + truncation: Optional[bool] = False, + logits_cache: bool = True, + max_length: Optional[int] = None, + device: Optional[str] = "cuda", + dtype: Optional[Union[str, torch.dtype]] = "auto", + softmax_dtype: Optional[Union[str, torch.dtype]] = None, + batch_size: Optional[Union[int, str]] = 1, + max_batch_size: Optional[int] = 64, + trust_remote_code: Optional[bool] = False, + use_fast_tokenizer: Optional[bool] = True, + add_bos_token: Optional[bool] = False, + prefix_token_id: Optional[int] = None, + # arguments used for splitting a model across GPUs naively. + # only used if `parallelize=True`. + parallelize: Optional[bool] = False, + max_memory_per_gpu: Optional[Union[int, str]] = None, + max_cpu_memory: Optional[Union[int, str]] = None, + offload_folder: Optional[Union[str, os.PathLike]] = "./offload", + # PEFT, delta weights and quantization options + peft: Optional[str] = None, + delta: Optional[str] = None, + autogptq: Optional[Union[bool, str]] = False, + gptqmodel: Optional[bool] = False, + gguf_file: Optional[str] = None, + **kwargs, + ) -> None: + super().__init__() + # optionally: take in an already-initialized transformers.PreTrainedModel + if not isinstance(pretrained, str): + eval_logger.warning( + "`pretrained` model kwarg is not of type `str`. Many other model arguments may be ignored. Please do not launch via accelerate or use `parallelize=True` if passing an existing model this way." + ) + assert not parallelize, ( + "`parallelize=True` is not compatible with passing pre-initialized model to `pretrained`" + ) + self._model = pretrained + self._device = self._model.device + self._config = self._model.config + gpus = 0 + + else: + assert isinstance(device, str) + assert isinstance(pretrained, str) + assert isinstance(batch_size, (int, str)) + + gpus = torch.cuda.device_count() + accelerator_kwargs = InitProcessGroupKwargs(timeout=timedelta(weeks=52)) + accelerator = Accelerator(kwargs_handlers=[accelerator_kwargs]) + if accelerator.num_processes > 1: + self.accelerator = accelerator + + if "npu" in accelerator.device.type: + gpus = torch.npu.device_count() + + # using one process with no model parallelism + if not (parallelize or accelerator.num_processes > 1): + # use user-passed device + device_list = set( + ["cuda", "cpu"] + + [f"cuda:{i}" for i in range(gpus)] + + ["mps", "mps:0"] + + [f"npu:{i}" for i in range(gpus)] + ) + if device and device in device_list: + self._device = torch.device(device) + eval_logger.info(f"Using device '{device}'") + if device in ("mps", "mps:0") and version.parse( + torch.__version__ + ) < version.parse("2.1"): + raise RuntimeError( + f"mps requires torch >= 2.1. You have {torch.__version__}" + ) + else: + eval_logger.info("Device not specified") + eval_logger.info(f"Cuda Available? {torch.cuda.is_available()}") + self._device = ( + torch.device("cuda") + if torch.cuda.is_available() + else torch.device("cpu") + ) + else: # Parallelism managed by accelerate + if device != "cuda": + eval_logger.info( + f"Using `accelerate launch` or `parallelize=True`, device '{device}' will be overridden when placing model." + ) + # TODO: include in warning that `load_in_8bit` etc. affect this too + self._device = ( + self.accelerator.device + if hasattr(self, "accelerator") + else torch.device(device) + ) + + revision = str(revision) # cast to string if not already one + # TODO: update this to be less of a hack once subfolder is fixed in HF + revision = revision + ("/" + subfolder if subfolder is not None else "") + + self._get_config( + pretrained, + revision=revision, + trust_remote_code=trust_remote_code, + gguf_file=gguf_file, + ) + + # determine which of 'causal' and 'seq2seq' backends to use for HF models + self._get_backend( + config=self.config, backend=backend, trust_remote_code=trust_remote_code + ) + + # load tokenizer so we know tokenizer vocabulary size before loading model and PEFT + self._create_tokenizer( + pretrained, + tokenizer, + revision=revision, + trust_remote_code=trust_remote_code, + use_fast_tokenizer=use_fast_tokenizer, + gguf_file=gguf_file, + add_bos_token=add_bos_token, + ) + + # if we passed `pretrained` as a string, initialize our model now + if isinstance(pretrained, str): + self._create_model( + pretrained=pretrained, + revision=revision, + dtype=dtype, + trust_remote_code=trust_remote_code, + parallelize=parallelize, + gpus=gpus, + max_memory_per_gpu=max_memory_per_gpu, + max_cpu_memory=max_cpu_memory, + offload_folder=offload_folder, + peft=peft, + delta=delta, + autogptq=autogptq, + gptqmodel=gptqmodel, + gguf_file=gguf_file, + quantization_config=getattr(self.config, "quantization_config", None), + **kwargs, + ) + + # access self._model through self.model property outside this method + if isinstance(self.model, torch.nn.Module): + self.model.eval() + self.model.tie_weights() + + self.truncation = truncation + self.logits_cache = logits_cache + self.vocab_size = self.tokenizer.vocab_size + # select (or create) a pad token to use + self.tokenizer = configure_pad_token(self.tokenizer, model_config=self.config) + + self.add_bos_token = add_bos_token + if "gemma" in getattr(self.config, "model_type", ""): + self.add_bos_token = True + eval_logger.info( + f"Model type is '{self.config.model_type}', part of the Gemma family--a BOS token will be used as Gemma underperforms without it." + ) + + self._max_length = max_length + self.pretrained = pretrained + self.delta = delta + self.peft = peft + self.revision = revision + self.batch_schedule = 1 + self.batch_sizes = {} + self.max_batch_size = max_batch_size + self.softmax_dtype = ( + get_dtype(softmax_dtype) if softmax_dtype is not None else None + ) + + if str(batch_size).startswith("auto"): + batch_size = batch_size.split(":") + self.batch_size_per_gpu = batch_size[0] + self.batch_schedule = float(batch_size[1]) if len(batch_size) > 1 else 1 + else: + self.batch_size_per_gpu = int(batch_size) + + if isinstance(pretrained, str): + if gpus >= 1 or str(self.device) == "mps": + # TODO: can remove this whole snippet except in the mps case, perhaps? + if not (parallelize or autogptq or hasattr(self, "accelerator")): + # place model onto device requested manually, + # if not using HF Accelerate or device_map + # or any other option that preloads model onto device + try: + self.model.to(self.device) + except ValueError: + eval_logger.debug( + "Failed to place model onto specified device. This may be because the model is quantized via `bitsandbytes` or `device_map` is provided. If the desired GPU is being used, this message is safe to ignore." + ) + # multigpu data-parallel support when launched with accelerate + if gpus > 1: + if accelerator.num_processes > 1: + if parallelize: + eval_logger.warning( + "You are both using a HF Accelerate `device_map` (`--model_args parallelize=True`) and launching via `accelerate launch`. This will attempt to do model and data parallelism depending on the resources available." + ) + elif gpus > accelerator.num_processes: + eval_logger.warning( + "WARNING: The number of total system GPUs does not match the number of spawned processes. " + "If you would like to use data parallelism, please launch the script " + "with 'accelerate launch *script*'. " + f"Current run will proceed with {accelerator.num_processes} devices." + ) + if self.accelerator.is_local_main_process: + eval_logger.info( + f"Using {gpus} devices with data parallelism" + ) + + self._device = torch.device(f"{accelerator.device}") + self.accelerator = accelerator + + self._rank = self.accelerator.local_process_index + self._world_size = self.accelerator.num_processes + else: + # if we aren't launching via accelerate, ditch + self._rank = 0 + self._world_size = 1 + else: + # if a PreTrainedModel was passed into HFLM, we forgo distributed setup. + eval_logger.warning( + "Passed an already-initialized model through `pretrained`, assuming single-process call to evaluate() or custom distributed integration" + ) + self._rank = 0 + self._world_size = 1 + + self.custom_prefix_token_id = prefix_token_id + if prefix_token_id is not None: + eval_logger.info( + f"Loglikelihood prefix token id used in evaluation: {self.prefix_token_id}" + ) + + def _get_accelerate_args( + self, + parallelize: Optional[bool] = None, + device_map: Optional[str] = "auto", + max_memory_per_gpu: Optional[Union[int, str]] = None, + max_cpu_memory: Optional[Union[int, str]] = None, + offload_folder: Optional[str] = "./offload", + gpus: Optional[int] = None, + ) -> dict: + """Returns the kwargs needed to apply `accelerate` in `AutoModel.from_pretrained`.""" + num_local_processes = int(os.environ.get("LOCAL_WORLD_SIZE", 1)) + num_machines = int(os.environ.get("WORLD_SIZE", 0)) // num_local_processes + if ( + num_machines == 0 + and hasattr(self, "accelerator") + and self.accelerator is not None + ): + eval_logger.info( + "We are not in a distributed setting for accelerate. Setting model_parallel to False." + ) + parallelize = False + + if parallelize is None: + # If parallelism is unset by the user, we automatically assign model parallelism + # if enough extra GPUs are available + max_memory_all_gpus = get_max_memory() + # We just want gpu, not cpu, max memory + if "cpu" in max_memory_all_gpus: + del max_memory_all_gpus["cpu"] + parallelize = bool(num_local_processes < len(max_memory_all_gpus)) + eval_logger.info( + f"Setting model parallel to {parallelize} since " + f"the number of local processes is {num_local_processes} " + f"and the number of GPUs is {len(max_memory_all_gpus)}" + ) + + args = {} + if parallelize: # Model parallelism will be used + max_memory = {} + if max_memory_per_gpu is not None: # Using the provided memory requirements + max_memory_per_gpu_map = { + device_idx: max_memory_per_gpu for device_idx in range(gpus) + } + else: # Estimating the possible memory requirements + max_memory_all_gpus = get_max_memory() + if "cpu" in max_memory_all_gpus: + del max_memory_all_gpus["cpu"] + if not hasattr(self, "accelerator"): + max_memory_per_gpu_map = { + k: v for k, v in max_memory_all_gpus.items() + } + else: + # use only 1 / num_processes of the GPUs if we are running under accelerate launch + max_memory_per_gpu_map = { + k: v + for k, v in max_memory_all_gpus.items() + if k % num_local_processes + == (self.accelerator.process_index % num_local_processes) + } + args["max_memory"] = max_memory_per_gpu_map + args["device_map"] = "auto" if device_map is None else device_map + eval_logger.info( + f"Model parallel was set to True, setting max memory per GPU to {max_memory_per_gpu_map} and device map to {args.get('device_map')}" + ) + + if max_cpu_memory is not None: + max_memory["cpu"] = max_cpu_memory + + args["offload_folder"] = offload_folder + elif ( + device_map is None + ): # No model parallelism, we use the default provided device for our model + if hasattr(self, "accelerator"): + device_map = {"": f"{self.accelerator.device}"} + else: + device_map = {"": str(self.device)} + args["max_memory"] = None + args["device_map"] = device_map + eval_logger.info( + f"Model parallel was set to False, max memory was not set, and device map was set to {device_map}" + ) + else: + args["max_memory"] = None + args["device_map"] = None + eval_logger.info("Model parallel was set to False.") + + return args + + @property + def config(self): + # return the associated transformers.AutoConfig for the given pretrained model. + return self._config + + @property + def model(self): + # returns the model, unwrapping it if using Accelerate + if hasattr(self, "accelerator"): + return self.accelerator.unwrap_model(self._model) + else: + return self._model + + @property + def eot_token_id(self): + # we use EOT because end of *text* is more accurate for what we're doing than end of *sentence* + return self.tokenizer.eos_token_id + + @property + def prefix_token_id(self): + # it is used as prefix for loglikelihood + if self.custom_prefix_token_id is not None: + return self.custom_prefix_token_id + if self.tokenizer.bos_token_id is not None: + return self.tokenizer.bos_token_id + return self.tokenizer.eos_token_id + + @property + def max_length(self): + if self._max_length: # if max length manually set, return it + return self._max_length + seqlen_config_attrs = ("n_positions", "max_position_embeddings", "n_ctx") + for attr in seqlen_config_attrs: + if hasattr(self.model.config, attr): + return getattr(self.model.config, attr) + if hasattr(self.tokenizer, "model_max_length"): + if self.tokenizer.model_max_length == 1000000000000000019884624838656: + return self._DEFAULT_MAX_LENGTH + return self.tokenizer.model_max_length + return self._DEFAULT_MAX_LENGTH + + @property + def max_gen_toks(self) -> int: + return 256 + + @property + def batch_size(self): + return self.batch_size_per_gpu + + @property + def device(self): + return self._device + + @property + def rank(self): + return self._rank + + @property + def world_size(self): + return self._world_size + + @property + def tokenizer_name(self) -> str: + return self.tokenizer.name_or_path.replace("/", "__") + + def _get_backend( + self, + config: Union[transformers.PretrainedConfig, transformers.AutoConfig], + backend: Literal["default", "causal", "seq2seq"] = "default", + trust_remote_code: Optional[bool] = False, + ) -> None: + """ + Helper method during initialization. + Determines the backend ("causal" (decoder-only) or "seq2seq" (encoder-decoder)) model type to be used. + sets `self.AUTO_MODEL_CLASS` appropriately if not already set. + + **If not calling HFLM.__init__() or HFLM._get_backend() within a subclass of HFLM, + user must set `self.backend` to be either "causal" or "seq2seq" manually!** + """ + + assert backend in ["default", "causal", "seq2seq"] + + if backend != "default": + # if we've settled on non-default backend, use that manually + if backend == "causal": + self.backend = backend + elif backend == "seq2seq": + self.backend = backend + eval_logger.info( + f"Overrode HF model backend type, and using type '{self.backend}'" + ) + else: + # determine and use the default HF backend for this model, based on its config + metadata. + if ( + getattr(config, "model_type") + in MODEL_FOR_SEQ_TO_SEQ_CAUSAL_LM_MAPPING_NAMES + ): + # first check if model type is listed under seq2seq models, since some + # models like MBart are listed in both seq2seq and causal mistakenly in HF transformers. + # these special cases should be treated as seq2seq models. + self.backend = "seq2seq" + eval_logger.debug(f"Using model type '{self.backend}'") + elif ( + getattr(self.config, "model_type") in MODEL_FOR_CAUSAL_LM_MAPPING_NAMES + ): + self.backend = "causal" + eval_logger.debug(f"Using model type '{self.backend}'") + else: + if not trust_remote_code: + eval_logger.warning( + "HF model type is neither marked as CausalLM or Seq2SeqLM. \ + This is expected if your model requires `trust_remote_code=True` but may be an error otherwise." + "Setting backend to causal" + ) + # if model type is neither in HF transformers causal or seq2seq model registries + # then we default to assuming AutoModelForCausalLM + self.backend = "causal" + eval_logger.info( + f"Model type cannot be determined. Using default model type '{self.backend}'" + ) + + if self.AUTO_MODEL_CLASS is None: + if self.backend == "causal": + self.AUTO_MODEL_CLASS = transformers.AutoModelForCausalLM + elif self.backend == "seq2seq": + self.AUTO_MODEL_CLASS = transformers.AutoModelForSeq2SeqLM + + def _get_config( + self, + pretrained: str, + revision: str = "main", + trust_remote_code: bool = False, + gguf_file: Optional[str] = None, + ) -> None: + """Return the model config for HuggingFace models""" + self._config = transformers.AutoConfig.from_pretrained( + pretrained, + revision=revision, + trust_remote_code=trust_remote_code, + gguf_file=gguf_file, + ) + + def _create_model( + self, + pretrained: str, + revision: Optional[str] = "main", + dtype: Optional[Union[str, torch.dtype]] = "auto", + trust_remote_code: Optional[bool] = False, + # arguments used for splitting a model across GPUs naively. + # only used if `parallelize=True`. + # (accelerate naive PP (device_map) options) + parallelize: Optional[bool] = False, + gpus: Optional[int] = None, + max_memory_per_gpu: Optional[Union[int, str]] = None, + max_cpu_memory: Optional[Union[int, str]] = None, + offload_folder: Optional[str] = "./offload", + # PEFT, delta weights and quantization options + peft: Optional[str] = None, + delta: Optional[str] = None, + autogptq: Optional[Union[bool, str]] = False, + gptqmodel: Optional[bool] = False, + gguf_file: Optional[str] = None, + quantization_config: Optional[Dict[str, Any]] = None, + **kwargs, + ) -> None: + """ + Initializes an HF or HF-compatible PreTrainedModel from scratch + inside HFLM, using the kwargs passed into self.__init__(). + + Also handles functionality such as AutoGPTQ usage and PEFT wrapping. + + For future similar extensions to AutoGPTQ that are not core to HF's ecosystem, + (such as PyTorch models that are nearly, but not quite, fully mirroring + HF's public interface relied on in this HFLM class) + please consider subclassing HFLM and overriding this and other methods as needed. + """ + + model_kwargs = kwargs if kwargs else {} + + model_kwargs.update( + self._get_accelerate_args( + parallelize=parallelize, + device_map=kwargs.get("device_map", None), + max_memory_per_gpu=max_memory_per_gpu, + max_cpu_memory=max_cpu_memory, + offload_folder=offload_folder, + gpus=gpus, + ) + ) + + if not autogptq and not gptqmodel: + if model_kwargs.get("load_in_4bit", None): + assert transformers.__version__ >= "4.30.0", ( + "load_in_4bit requires transformers >= 4.30.0" + ) + if transformers.__version__ >= "4.30.0": + if model_kwargs.get("load_in_4bit", None): + if model_kwargs.get("bnb_4bit_compute_dtype", None): + model_kwargs["bnb_4bit_compute_dtype"] = get_dtype( + model_kwargs["bnb_4bit_compute_dtype"] + ) + + self._model = self.AUTO_MODEL_CLASS.from_pretrained( + pretrained, + revision=revision, + torch_dtype=get_dtype(dtype), + trust_remote_code=trust_remote_code, + gguf_file=gguf_file, + quantization_config=quantization_config, + **model_kwargs, + ) + else: + if autogptq and gptqmodel: + raise ValueError( + "Cannot use both 'autogptq' and 'gptqmodel' options at the same time." + ) + + if autogptq: + try: + from auto_gptq import AutoGPTQForCausalLM + except ModuleNotFoundError as exception: + raise type(exception)( + "Tried to load auto_gptq, but auto-gptq is not installed ", + "please install auto-gptq via pip install lm-eval[gptq] or pip install -e .[gptq]", + ) + + self._model = AutoGPTQForCausalLM.from_quantized( + pretrained, + trust_remote_code=trust_remote_code, + model_basename=None if autogptq is True else Path(autogptq).stem, + use_safetensors=True + if autogptq is True + else autogptq.endswith(".safetensors"), + **model_kwargs, + ) + + if gptqmodel: + try: + from gptqmodel import GPTQModel + except ModuleNotFoundError as exception: + raise type(exception)( + "Tried to load gptqmodel, but gptqmodel is not installed ", + "please install gptqmodel via `pip install gptqmodel --no-build-isolation` or `pip install lm-eval[gptqmodel] --no-build-isolation`", + ) + + self._model = GPTQModel.from_quantized( + pretrained, trust_remote_code=trust_remote_code, **model_kwargs + ) + + if peft and delta: + raise ValueError( + "Cannot use both 'peft' and 'delta' options at the same time." + ) + + if peft: + if model_kwargs.get("load_in_4bit", None): + if version.parse(PEFT_VERSION) < version.parse("0.4.0"): + raise AssertionError("load_in_4bit requires peft >= 0.4.0") + if self._model.config.vocab_size != len(self.tokenizer): + # resize model for LoRAs with added tokens + eval_logger.info( + f"Model config indicates vocab_size='{self._model.config.vocab_size}', but found tokenizer with vocab size '{len(self.tokenizer)}'. Resizing model embedding layer..." + ) + self._model.resize_token_embeddings(len(self.tokenizer)) + self._model = PeftModel.from_pretrained( + self._model, peft, revision=revision + ) + elif delta: + if autogptq: + eval_logger.warning( + "Delta weights might trigger unexpected behavior when used with AutoGPTQ." + ) + _model_delta = self.AUTO_MODEL_CLASS.from_pretrained( + delta, + revision=revision, + torch_dtype=get_dtype(dtype), + trust_remote_code=trust_remote_code, + **model_kwargs, + ) + for name, param in self._model.state_dict().items(): + try: + param.data += _model_delta.state_dict()[name] + except KeyError: + raise KeyError(f"Delta model is missing weights for layer: {name}") + except Exception as e: + raise RuntimeError( + f"Failed to add delta weights to layer {name}. Error: {e}" + ) + + del _model_delta + + return None + + def _create_tokenizer( + self, + pretrained: Union[str, transformers.PreTrainedModel], + tokenizer: Optional[ + Union[ + str, + transformers.PreTrainedTokenizer, + transformers.PreTrainedTokenizerFast, + ] + ], + revision: Optional[str] = "main", + trust_remote_code: Optional[bool] = False, + use_fast_tokenizer: Optional[bool] = True, + gguf_file: Optional[str] = None, + add_bos_token: Optional[bool] = False, + ) -> None: + """ + Helper method during initialization. + + Create a tokenizer object corresponding to the correct + tokenizer for value of `pretrained`, or use the pre-initialized tokenizer passed. + """ + kwargs = { + "revision": revision, + "trust_remote_code": trust_remote_code, + } + + # gguf format embeds tokenizer and is not compatible with hf tokenizer `use_fast` param + if gguf_file is not None: + kwargs["gguf_file"] = gguf_file + else: + kwargs["use_fast"] = use_fast_tokenizer + + if add_bos_token: + kwargs["add_bos_token"] = True + + if tokenizer: + if isinstance(tokenizer, str): + self.tokenizer = transformers.AutoTokenizer.from_pretrained( + tokenizer, **kwargs + ) + else: + assert isinstance( + tokenizer, transformers.PreTrainedTokenizer + ) or isinstance(tokenizer, transformers.PreTrainedTokenizerFast) + self.tokenizer = tokenizer + else: + # Get tokenizer based on 'pretrained' + if isinstance(pretrained, str): + model_name = pretrained + else: + # get the HF hub name via accessor on model + model_name = self.model.name_or_path + self.tokenizer = transformers.AutoTokenizer.from_pretrained( + model_name, **kwargs + ) + return None + + def _detect_batch_size(self, requests=None, pos: int = 0): + if requests: + _, context_enc, continuation_enc = requests[pos] + max_length = len( + (context_enc + continuation_enc)[-(self.max_length + 1) :][:-1] + ) + max_context_enc = len(context_enc[-(self.max_length + 1) :]) + max_cont_enc = len(continuation_enc[-(self.max_length + 1) :]) + else: + max_length = self.max_length + max_context_enc = max_length + max_cont_enc = max_length + + # if OOM, then halves batch_size and tries again + @find_executable_batch_size(starting_batch_size=self.max_batch_size) + def forward_batch(batch_size): + if self.backend == "seq2seq": + length = max(max_context_enc, max_cont_enc) + batched_conts = torch.ones( + (batch_size, length), device=self.device + ).long() + test_batch = torch.ones((batch_size, length), device=self.device).long() + call_kwargs = { + "attn_mask": test_batch, + "labels": batched_conts, + } + else: + call_kwargs = {} + test_batch = torch.ones( + (batch_size, max_length), device=self.device + ).long() + for _ in range(5): + out = F.log_softmax( # noqa: F841 + self._model_call(test_batch, **call_kwargs), + dim=-1, + dtype=self.softmax_dtype, + ) + + return batch_size + + try: + batch_size = forward_batch() + except RuntimeError as e: + if "No executable batch size found" in str(e): + batch_size = 1 + else: + raise + + if self.world_size > 1: + # if multi-GPU, always take minimum over all selected batch sizes + max_rnk_bs = torch.tensor([batch_size], device=self.device) + gathered = ( + self.accelerator.gather(max_rnk_bs).cpu().detach().numpy().tolist() + ) + batch_size = min(gathered) + clear_torch_cache() + return batch_size + + clear_torch_cache() + return batch_size + + def tok_encode( + self, string: str, left_truncate_len=None, add_special_tokens=None + ) -> List[int]: + """ """ + # default for None - empty dict, use predefined tokenizer param + # used for all models except for CausalLM or predefined value + special_tokens_kwargs = {} + + # by default for CausalLM - false or self.add_bos_token is set + if add_special_tokens is None: + if self.backend == "causal": + special_tokens_kwargs = { + "add_special_tokens": False or self.add_bos_token + } + # otherwise the method explicitly defines the value + else: + special_tokens_kwargs = {"add_special_tokens": add_special_tokens} + + encoding = self.tokenizer.encode(string, **special_tokens_kwargs) + + # left-truncate the encoded context to be at most `left_truncate_len` tokens long + if left_truncate_len: + encoding = encoding[-left_truncate_len:] + + return encoding + + def tok_batch_encode( + self, + strings: List[str], + padding_side: str = "left", + left_truncate_len: int = None, + truncation: bool = False, + ) -> Tuple[torch.Tensor, torch.Tensor]: + # encode a batch of strings. converts to tensors and pads automatically, unlike tok_encode. + old_padding_side = self.tokenizer.padding_side + self.tokenizer.padding_side = padding_side + + add_special_tokens = {} + if self.backend == "causal": + add_special_tokens = {"add_special_tokens": False or self.add_bos_token} + + encoding = self.tokenizer( + strings, + truncation=truncation, + padding="longest", + return_tensors="pt", + **add_special_tokens, + ) + if left_truncate_len: + original_lengths = encoding["input_ids"].size(1) + if original_lengths > left_truncate_len: + eval_logger.warn( + f"Left truncation applied. Original sequence length was {original_lengths}, " + f"truncating to last {left_truncate_len} tokens. Some content will be lost.", + ) + encoding["input_ids"] = encoding["input_ids"][:, -left_truncate_len:] + encoding["attention_mask"] = encoding["attention_mask"][ + :, -left_truncate_len: + ] + self.tokenizer.padding_side = old_padding_side + + return encoding["input_ids"], encoding["attention_mask"] + + def tok_decode(self, tokens, skip_special_tokens=True): + return self.tokenizer.decode(tokens, skip_special_tokens=skip_special_tokens) + + def _model_call(self, inps, attn_mask=None, labels=None): + """ + :param inps: torch.Tensor + A torch tensor of shape [batch, (sequence_ctx + sequence_cont)] or of shape + [batch, sequence_ctx]. the size of sequence may vary from call to call + :param attn_mask: torch.Tensor, optional + A torch tensor of shape [batch, (sequence_ctx + sequence_cont)]. Only passed + (and must be passed) if self.AUTO_MODEL_CLASS is transformers.AutoModelForSeq2SeqLM + :param labels: torch.Tensor, optional + A torch tensor of shape [batch, (sequence_ctx + sequence_cont)]. Only passed + (and must be passed) if self.AUTO_MODEL_CLASS is transformers.AutoModelForSeq2SeqLM + :return + A torch tensor of shape [batch, sequence, vocab] with the + logits returned from the model's decoder + """ + with torch.no_grad(): + if attn_mask is not None or labels is not None: + assert attn_mask is not None and labels is not None + assert self.AUTO_MODEL_CLASS == transformers.AutoModelForSeq2SeqLM + return self.model( + input_ids=inps, attention_mask=attn_mask, labels=labels + ).logits + else: + assert self.AUTO_MODEL_CLASS in ( + transformers.AutoModelForCausalLM, + transformers.AutoModelForVision2Seq, + ) + return self.model(inps).logits + + def _model_generate(self, context, max_length, stop, **generation_kwargs): + # temperature = 0.0 if not set + # if do_sample is false and temp==0.0: + # remove temperature, as do_sample=False takes care of this + # and we don't want a warning from HF + generation_kwargs["temperature"] = generation_kwargs.get("temperature", 0.0) + do_sample = generation_kwargs.get("do_sample", None) + + # The temperature has to be a strictly positive float -- if it is 0.0, use greedy decoding strategies + if generation_kwargs.get("temperature") == 0.0 and do_sample is None: + generation_kwargs["do_sample"] = do_sample = False + + if do_sample is False and generation_kwargs.get("temperature") == 0.0: + generation_kwargs.pop("temperature") + # build stopping criteria + stopping_criteria = stop_sequences_criteria( + self.tokenizer, stop, context.shape[1], context.shape[0] + ) + return self.model.generate( + input_ids=context, + max_length=max_length, + stopping_criteria=stopping_criteria, + pad_token_id=self.tokenizer.pad_token_id, + use_cache=True, + **generation_kwargs, + ) + + def _select_cont_toks( + self, logits: torch.Tensor, contlen: int = None, inplen: int = None + ) -> torch.Tensor: + if self.backend == "causal": + assert contlen and inplen, ( + "Must pass input len and cont. len to select scored logits for causal LM" + ) + # discard right-padding. + # also discard the input/context tokens. we'll only score continuations. + logits = logits[inplen - contlen : inplen] + elif self.backend == "seq2seq": + assert contlen and not inplen, ( + "Selecting scored logits for Seq2SeqLM requires only cont. len" + ) + # only discard right-padding. + # the logits input to this fn only contain decoder-side tokens. + logits = logits[:contlen] + + return logits + + def loglikelihood_rolling( + self, requests: List[Instance], disable_tqdm: bool = False + ) -> List[float]: + adaptive_batch_size = None + if self.batch_size == "auto": + # using rolling window with maximum context + print("Passed argument batch_size = auto. Detecting largest batch size") + batch_size = self._detect_batch_size() + print(f"Determined Largest batch size: {batch_size}") + adaptive_batch_size = batch_size + + # First, collect all windows from all requests + all_windows = [] # List of (request_idx, window) tuples + request_window_counts = [] # Track number of windows per request + + for req_idx, (string,) in enumerate( + tqdm( + [req.args for req in requests], + disable=(disable_tqdm or (self.rank != 0)), + ) + ): + rolling_token_windows: List[Tuple[List[int], List[int]]] = list( + map( + utils.make_disjoint_window, + utils.get_rolling_token_windows( + token_list=self.tok_encode(string), + prefix_token=self.prefix_token_id, + max_seq_len=self.max_length, + context_len=1, + ), + ) + ) + + # TODO: Right now, we pass single EOT token to the Encoder and the full context to the decoder, in seq2seq case + windows = [(None,) + x for x in rolling_token_windows] + + # Store windows with their request index + all_windows.extend((req_idx, window) for window in windows) + request_window_counts.append(len(windows)) + + # Handle distributed case padding + pad_amnt = 0 + if self.world_size > 1: + mytensor = torch.tensor(len(all_windows), device=self.device) + gathered = self.accelerator.gather(mytensor).cpu().detach().numpy().tolist() + pad_amnt = max(gathered) - gathered[self.rank] + if pad_amnt > 0: + all_windows += pad_amnt * [all_windows[0]] + + all_nlls = [] + batch_size = adaptive_batch_size or self.batch_size + for i in range(0, len(all_windows), batch_size): + batch = all_windows[i : i + batch_size] + # Extract just the windows for processing, keeping track of request indices + batch_indices, batch_windows = zip(*batch) + + batch_nlls = self._loglikelihood_tokens( + requests=batch_windows, + disable_tqdm=False, + override_bs=len(batch_windows), + ) + # Store results with their request indices + all_nlls.extend(zip(batch_indices, batch_nlls)) + + # Remove padding if necessary + if (self.world_size > 1) and (pad_amnt > 0): + all_nlls = all_nlls[:-pad_amnt] + + # Reconstruct per-request loglikelihoods + loglikelihoods = [] + current_idx = 0 + for window_count in request_window_counts: + # Get all nlls for this request + request_nlls = all_nlls[current_idx : current_idx + window_count] + # Sum up the nlls for this request (discarding is_greedy) + request_total = sum(nll[0] for _, nll in request_nlls) + loglikelihoods.append(request_total) + current_idx += window_count + + string = requests[len(loglikelihoods) - 1].args[0] + self.cache_hook.add_partial( + "loglikelihood_rolling", (string,), request_total + ) + + return loglikelihoods + + def _batch_scheduler(self, pos, n_reordered_requests): + sched = pos // int(len(n_reordered_requests) / self.batch_schedule) + if sched in self.batch_sizes: + return self.batch_sizes[sched] + if (len(self.batch_sizes) > 1) and ( + self.batch_sizes[sched - 1] == self.max_batch_size + ): + # if previous batch size is already maximal, skip recomputation + self.batch_sizes[sched] = self.max_batch_size + return self.batch_sizes[sched] + print( + f"Passed argument batch_size = auto:{self.batch_schedule}. Detecting largest batch size" + ) + self.batch_sizes[sched] = self._detect_batch_size(n_reordered_requests, pos) + print(f"Determined largest batch size: {self.batch_sizes[sched]}") + return self.batch_sizes[sched] + + def _loglikelihood_tokens( + self, + requests: List[Tuple[Tuple[str, str], List[int], List[int]]], + disable_tqdm: bool = False, + override_bs: int = None, + ) -> List[Tuple[float, bool]]: + # TODO: implement some kind of efficient-request-middleware that lumps together requests with the same context + res = [] + + def _collate(req: Tuple[Tuple[str, str], List[int], List[int]]): + """Defines the key for the sorted method""" + # the negative sign on len(toks) sorts descending - this has a few advantages: + # - time estimates will always be over not underestimates, which is more useful for planning + # - to know the size of a batch when going through the list, you know the first one is always the batch + # padded context length. this is useful to simplify the batching logic and more importantly to make + # automatic adaptive batches much much easier to implement + # - any OOMs will happen right away rather than near the end + + toks = req[1] + req[2] + return -len(toks), tuple(toks) + + def _lookup_one_token_cont(req: Tuple[Tuple[str, str], List[int], List[int]]): + """Defines the key to group and lookup one-token continuations""" + # Use with group_by="contexts" (optional)" + # allows for the creation of a lookup, so we can reuse logits in case of one-token continuations. + # speeds up some multiple-choice tasks proportionally to the number of choices. + # groups requests by context+continuation[:-1] and infer on one request/group. + return req[-2] + req[-1][:-1] + + re_ord = Collator( + requests, + sort_fn=_collate, + group_by="contexts" + if self.backend == "causal" and self.logits_cache + else None, + group_fn=_lookup_one_token_cont, + ) + + # automatic (variable) batch size detection for vectorization + # pull longest context sample from request + n_reordered_requests = len(re_ord) + batch_size = ( + self.batch_size + if self.batch_size != "auto" + else override_bs + if override_bs is not None + else 0 + ) + batch_fn = ( + self._batch_scheduler + if self.batch_size == "auto" + and n_reordered_requests > 0 + and not override_bs + else None + ) + + chunks = re_ord.get_batched(n=batch_size, batch_fn=batch_fn) + pbar = tqdm( + total=len(requests), + disable=(disable_tqdm or (self.rank != 0)), + desc="Running loglikelihood requests", + ) + for chunk in chunks: + inps = [] + cont_toks_list = [] + inplens = [] + + conts = [] + encoder_attns = [] + + padding_len_inp = None + padding_len_cont = None + # because vectorizing is annoying, we first convert each (context, continuation) pair to padded + # tensors, then we pack them together into a batch, call the model, and then pick it all apart + # again because vectorizing is annoying + + for _, context_enc, continuation_enc in chunk: + # sanity check + assert len(context_enc) > 0 + assert len(continuation_enc) > 0 + assert len(continuation_enc) <= self.max_length + + # how this all works (illustrated on a causal decoder-only setup): + # CTX CONT + # inp 0 1 2 3|4 5 6 7 8 9 <- last token is deleted by inp[:, :-1] + # model \ \ + # logits 1 2 3|4 5 6 7 8 9 <- the ctx half gets tossed out by the + # cont_toks 4 5 6 7 8 9 [:, -len(continuation_enc):, :self.vocab_size] slice + + # when too long to fit in context, truncate from the left + if self.backend == "causal": + total_length = len(context_enc) + len(continuation_enc) + if total_length > self.max_length + 1: + eval_logger.warning( + f"Combined length of context ({len(context_enc)}) and continuation ({len(continuation_enc)}) " + f"exceeds model's maximum length ({self.max_length}). " + f"Truncating {total_length - self.max_length + 1} tokens from the left." + ) + inp = torch.tensor( + (context_enc + continuation_enc)[-(self.max_length + 1) :][:-1], + dtype=torch.long, + device=self.device, + ) + (inplen,) = inp.shape + elif self.backend == "seq2seq": + inp = torch.tensor( + (context_enc)[-self.max_length :], + dtype=torch.long, + device=self.device, + ) + (inplen,) = inp.shape + + # build encoder attn masks + encoder_attns.append(torch.ones_like(inp)) + + cont = torch.tensor( + (continuation_enc)[-self.max_length :], + # TODO: left-shift these? + # TODO: our code assumes we never end up truncating conts for either model type + dtype=torch.long, + device=self.device, + ) + (contlen,) = cont.shape + + conts.append(cont) + + padding_len_cont = ( + max(padding_len_cont, contlen) + if padding_len_cont is not None + else contlen + ) + + padding_len_inp = ( + max(padding_len_inp, inplen) + if padding_len_inp is not None + else inplen + ) + + inps.append(inp) # [1, inp_length] + cont_toks_list.append(continuation_enc) + inplens.append(inplen) + + # create encoder attn mask and batched conts, if seq2seq + call_kwargs = {} + if self.backend == "causal": + batched_inps = pad_and_concat( + padding_len_inp, inps, padding_side="right" + ) # [batch, padding_len_inp] + elif self.backend == "seq2seq": + # TODO: left-pad encoder inps and mask? + batched_inps = pad_and_concat( + padding_len_inp, inps + ) # [batch, padding_len_inp] + batched_conts = pad_and_concat( + padding_len_cont, conts + ) # [batch, padding_len_cont] + batched_encoder_mask = pad_and_concat( + padding_len_inp, encoder_attns + ) # [batch, padding_len_inp] + call_kwargs = { + "attn_mask": batched_encoder_mask, + "labels": batched_conts, + } + + multi_logits = F.log_softmax( + self._model_call(batched_inps, **call_kwargs), + dim=-1, + dtype=self.softmax_dtype, + ) # [batch, padding_length (inp or cont), vocab] + + for (request_str, ctx_tokens, _), logits, inplen, cont_toks in zip( + chunk, multi_logits, inplens, cont_toks_list + ): + # Slice to original seq length + contlen = len(cont_toks) + # take only logits in the continuation + # (discard context toks if decoder-only ; discard right-padding) + # also discards + checks for "virtual tokens" in the causal LM's input window + # from prompt/prefix tuning tokens, if applicable + ctx_len = ( + inplen + (logits.shape[0] - padding_len_inp) + if self.backend == "causal" + else None + ) + logits = self._select_cont_toks(logits, contlen=contlen, inplen=ctx_len) + logits = logits.unsqueeze(0) # [1, seq, vocab] + + # Check if per-token argmax is exactly equal to continuation + greedy_tokens = logits.argmax(dim=-1) + + # check for one-token continuation cache hits. + # noop in case group_by != "contexts" or no cache hit and returns the + # original args. Otherwise, expands the logits batch dimension and yields each + # batch along with matching continuation tokens and prompt strings. + # logits -> [1, seq, vocab] + for request_str, cont_toks, logits in re_ord.get_cache( + req_str=request_str, + cxt_toks=ctx_tokens, + cont_toks=cont_toks, + logits=logits, + ): + cont_toks = torch.tensor( + cont_toks, dtype=torch.long, device=self.device + ).unsqueeze(0) # [1, seq] + # Use trailing slice [-cont_toks.shape[1]:] to handle variable length cont_len (but same ctx+cont[:-1]). + # i.e. continuations can be sliced at diff points. Collator ensures we have sufficient greedy_tokens + # by choosing key with longest cont if group_by="contexts". + max_equal = ( + greedy_tokens[:, -cont_toks.shape[1] :] == cont_toks + ).all() + + # Obtain log-probs at the corresponding continuation token indices + # last_token_slice = logits[:, -1, :].squeeze(0).tolist() + logits = torch.gather(logits, 2, cont_toks.unsqueeze(-1)).squeeze( + -1 + ) # [1, seq] + + # Answer: (log prob, is-exact-match) + answer = (float(logits.sum()), bool(max_equal)) + + res.append(answer) + + if request_str is not None: + # special case: loglikelihood_rolling produces a number of loglikelihood requests + # all with cache key None. instead do add_partial on the per-example level + # in the loglikelihood_rolling() function for those. + self.cache_hook.add_partial( + "loglikelihood", request_str, answer + ) + pbar.update(1) + + pbar.close() + + return re_ord.get_original(res) + + def generate_until( + self, requests: List[Instance], disable_tqdm: bool = False + ) -> List[str]: + res = [] + + def _collate(req: Tuple[str, dict]): + """Defines the key for the sorted method""" + # the negative sign on len(toks) sorts descending - this has a few advantages: + # - time estimates will always be over not underestimates, which is more useful for planning + # - to know the size of a batch when going through the list, you know the first one is always the batch + # padded context length. this is useful to simplify the batching logic and more importantly to make + # automatic adaptive batches much much easier to implement + # - any OOMs will happen right away rather than near the end + toks = self.tok_encode(req[0]) + return -len(toks), req[0] + + pbar = tqdm( + total=len(requests), + disable=(disable_tqdm or (self.rank != 0)), + desc="Running generate_until requests", + ) + adaptive_batch_size = None + if self.batch_size == "auto": + # using rolling window with maximum context + print("Passed argument batch_size = auto. Detecting largest batch size") + batch_size = self._detect_batch_size() + print(f"Determined Largest batch size: {batch_size}") + adaptive_batch_size = batch_size + # for each different set of kwargs, we execute all requests, by batch. + batch_size = ( + self.batch_size + if self.batch_size != "auto" + else adaptive_batch_size + if adaptive_batch_size is not None + else 0 + ) + batch_fn = ( + self._batch_scheduler + if self.batch_size == "auto" and not adaptive_batch_size + else None + ) + + # we group requests by their generation_kwargs, + # so that we don't try to execute e.g. greedy sampling and temp=0.8 sampling + # in the same batch. + # group_fn=lambda x: x[1] -> x=(context, gen_kwargs) + re_ords = Collator( + [reg.args for reg in requests], + sort_fn=_collate, + group_by="gen_kwargs", + group_fn=lambda x: x[1], + ) + chunks = re_ords.get_batched(n=batch_size, batch_fn=batch_fn) + eos = self.tok_decode(self.eot_token_id, skip_special_tokens=False) + for chunk in chunks: + contexts, all_gen_kwargs = zip(*chunk) + # we assume all gen kwargs in the batch are the same + # this is safe to assume because the `grouper` object ensures it. + gen_kwargs = all_gen_kwargs[0] + # unpack our keyword arguments. + if isinstance(gen_kwargs, dict): + kwargs = copy.deepcopy(gen_kwargs) # edge case for repeats > 1 + # add EOS token to stop sequences + until = handle_stop_sequences(kwargs.pop("until", None), eos=eos) + else: + raise ValueError( + f"Expected `kwargs` to be of type `dict` but got {type(gen_kwargs)}" + ) + if "max_gen_toks" in kwargs.keys(): + max_gen_toks = kwargs.pop("max_gen_toks") + else: + max_gen_toks = self.max_gen_toks + + # set the max length in tokens of inputs ("context_enc") + if self.backend == "causal": + # max len for inputs = max length, minus room to generate the max new tokens + max_ctx_len = self.max_length - max_gen_toks + assert max_ctx_len > 0, ( + f"Invalid configuration: requested max tokens to generate ({max_gen_toks}) must be less than model's maximum sequence length ({self.max_length})." + ) + elif self.backend == "seq2seq": + # max len for inputs = encoder's whole max_length + max_ctx_len = self.max_length + + # encode, pad, and truncate contexts for this batch + context_enc, attn_masks = self.tok_batch_encode( + contexts, + left_truncate_len=max_ctx_len, + truncation=self.truncation, + ) + context_enc = context_enc.to(self.device) + attn_masks = attn_masks.to(self.device) + + if "max_length" not in kwargs: + kwargs["max_length"] = context_enc.shape[1] + max_gen_toks + + # perform batched generation + cont = self._model_generate( + context=context_enc, + attention_mask=attn_masks, + stop=until, + **kwargs, + ) + + cont_toks_list = cont.tolist() + for cont_toks, context in zip(cont_toks_list, contexts): + # discard context + left-padding toks if using causal decoder-only LM + if self.backend == "causal": + cont_toks = cont_toks[context_enc.shape[1] :] + + s = self.tok_decode(cont_toks) + + # use secondary stop seqs to cut off should-have-been-stopped content post-hoc + for term in until: + if len(term) > 0: + # ignore '' separator, + # for seq2seq case where self.tok_decode(self.eot_token_id) = '' + s = s.split(term)[0] + + res.append(s) + + self.cache_hook.add_partial("generate_until", (context, gen_kwargs), s) + pbar.update(1) + # reorder this group of results back to original unsorted form + res = re_ords.get_original(res) + + pbar.close() + + return res + + def apply_chat_template( + self, chat_history: List[Dict[str, str]], add_generation_prompt: bool = True + ) -> str: + """ + Method to apply a chat template to a list of chat history between user and model. + """ + try: + chat_templated = self.tokenizer.apply_chat_template( + chat_history, + tokenize=False, + add_generation_prompt=add_generation_prompt, + continue_final_message=not add_generation_prompt, + ) + except jinja2.exceptions.TemplateError: + eval_logger.warning( + "Failed to apply chat template. removing the system role in chat history." + ) + chat_history = [msg for msg in chat_history if msg["role"] != "system"] + chat_templated = self.tokenizer.apply_chat_template( + chat_history, + tokenize=False, + add_generation_prompt=add_generation_prompt, + continue_final_message=not add_generation_prompt, + ) + + return chat_templated + + def get_model_info(self) -> dict: + """ + Method to get Hugging Face model information for experiment reproducibility. + """ + + def get_model_num_params(model) -> int: + if hasattr(model, "num_parameters"): + return model.num_parameters() + if hasattr(model, "parameters"): + return sum(p.numel() for p in model.parameters()) + else: + return -1 + + def get_model_dtype(model) -> str: + if hasattr(model, "dtype"): + return model.dtype + else: + return "" + + def get_model_sha(pretrained: str, revision: str) -> str: + try: + model_info = HfApi().model_info(repo_id=pretrained, revision=revision) + return model_info.sha + except Exception as e: + eval_logger.debug( + f"Failed to get model SHA for {pretrained} at revision {revision}. Error: {e}" + ) + return "" + + model_info = { + "model_num_parameters": get_model_num_params(self._model), + "model_dtype": get_model_dtype(self._model), + "model_revision": self.revision, + "model_sha": get_model_sha(self.pretrained, self.revision), + } + if self.peft: + model_info["peft_sha"] = get_model_sha(self.peft, self.revision) + if self.delta: + model_info["delta_sha"] = get_model_sha(self.delta, self.revision) + return model_info diff --git a/lm-evaluation-harness/lm_eval/models/optimum_ipex.py b/lm-evaluation-harness/lm_eval/models/optimum_ipex.py new file mode 100644 index 0000000000000000000000000000000000000000..e4753ff7176def55aabd080193fd36234275f625 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/models/optimum_ipex.py @@ -0,0 +1,79 @@ +import logging +from importlib.util import find_spec + +from lm_eval.api.registry import register_model +from lm_eval.models.huggingface import HFLM +from lm_eval.models.utils import get_dtype + + +eval_logger = logging.getLogger(__name__) + + +@register_model("ipex") +class IPEXLM(HFLM): + """ + using the HuggingFace transformers + optimum-intel ipex backend, can run on intel cpu and intel gpu + """ + + def __init__( + self, + **kwargs, + ) -> None: + if "backend" in kwargs: + # currently only supports causal models + assert kwargs["backend"] == "causal", ( + "Currently, only IPEXModelForCausalLM is supported." + ) + + super().__init__( + backend=kwargs.pop("backend", "causal"), + **kwargs, + ) + + def _create_model( + self, + pretrained: str, + revision="main", + dtype="auto", + trust_remote_code=False, + # arguments used for splitting a model across GPUs naively. + # only used if `parallelize=True`. + # (accelerate naive PP (device_map) options) + parallelize=False, + gpus=None, + max_memory_per_gpu=None, + max_cpu_memory=None, + offload_folder="./offload", + # PEFT, delta weights and quantization options + peft=None, + delta=None, + autogptq=False, + gptqmodel=False, + **kwargs, + ) -> None: + if not find_spec("optimum"): + raise ModuleNotFoundError( + "package `optimum` is not installed. Please install it via `pip install optimum[ipex]`" + ) + else: + from optimum.intel import IPEXModelForCausalLM + + model_kwargs = kwargs if kwargs else {} + model_kwargs.update( + self._get_accelerate_args( + parallelize=parallelize, + device_map=kwargs.get("device_map", None), + max_memory_per_gpu=max_memory_per_gpu, + max_cpu_memory=max_cpu_memory, + offload_folder=offload_folder, + gpus=gpus, + ) + ) + + self._model = IPEXModelForCausalLM.from_pretrained( + pretrained, + revision=revision, + torch_dtype=get_dtype(dtype), + trust_remote_code=trust_remote_code, + **model_kwargs, + ) diff --git a/lm-evaluation-harness/lm_eval/models/optimum_lm.py b/lm-evaluation-harness/lm_eval/models/optimum_lm.py new file mode 100644 index 0000000000000000000000000000000000000000..cce636ff10a6d7a8a0e7a8908f0c82a71c5b37ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/models/optimum_lm.py @@ -0,0 +1,92 @@ +import json +import logging +from importlib.util import find_spec +from pathlib import Path + +from lm_eval.api.registry import register_model +from lm_eval.models.huggingface import HFLM + + +eval_logger = logging.getLogger(__name__) + + +@register_model("openvino") +class OptimumLM(HFLM): + """ + Optimum Intel provides a simple interface to optimize Transformer models and convert them to \ + OpenVINO™ Intermediate Representation (IR) format to accelerate end-to-end pipelines on \ + Intel® architectures using OpenVINO™ runtime. + + To use an OpenVINO config, use `--model_args ov_config` to point to a json file with an OpenVINO config: + `lm_eval --model openvino --model_args pretrained=gpt2,ov_config=config.json --task lambada_openai` + Example json file contents: {"INFERENCE_PRECISION_HINT": "f32", "CACHE_DIR": "model_cache"} + """ + + def __init__( + self, + device="cpu", + **kwargs, + ) -> None: + if "backend" in kwargs: + # optimum currently only supports causal models + assert kwargs["backend"] == "causal", ( + "Currently, only OVModelForCausalLM is supported." + ) + + self.openvino_device = device + + super().__init__( + device=self.openvino_device, + backend=kwargs.pop("backend", "causal"), + **kwargs, + ) + + def _create_model( + self, + pretrained: str, + revision="main", + dtype="auto", + trust_remote_code=False, + **kwargs, + ) -> None: + if not find_spec("optimum"): + raise ModuleNotFoundError( + "package `optimum` is not installed. Please install it via `pip install optimum[openvino]`" + ) + else: + from optimum.intel.openvino import OVModelForCausalLM + + model_kwargs = kwargs if kwargs else {} + if "ov_config" in model_kwargs: + if not Path(model_kwargs["ov_config"]).exists(): + raise ValueError( + "ov_config should point to a .json file containing an OpenVINO config" + ) + with open(model_kwargs["ov_config"]) as f: + model_kwargs["ov_config"] = json.load(f) + eval_logger.info( + f"Using custom OpenVINO config: {model_kwargs['ov_config']}" + ) + + else: + model_kwargs["ov_config"] = {} + model_kwargs["ov_config"].setdefault("CACHE_DIR", "") + if "pipeline_parallel" in model_kwargs: + if model_kwargs["pipeline_parallel"]: + model_kwargs["ov_config"]["MODEL_DISTRIBUTION_POLICY"] = ( + "PIPELINE_PARALLEL" + ) + model_file = Path(pretrained) / "openvino_model.xml" + if model_file.exists(): + export = False + else: + export = True + + self._model = OVModelForCausalLM.from_pretrained( + pretrained, + revision=revision, + trust_remote_code=trust_remote_code, + export=export, + device=self.openvino_device.upper(), + **model_kwargs, + ) diff --git a/lm-evaluation-harness/lm_eval/models/textsynth.py b/lm-evaluation-harness/lm_eval/models/textsynth.py new file mode 100644 index 0000000000000000000000000000000000000000..a14f6287b6f11b21cfc69ca471bcbe99a631be12 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/models/textsynth.py @@ -0,0 +1,172 @@ +"""TextSynth API +Implementation provided by Fabrice Bellard: + https://github.com/EleutherAI/lm-evaluation-harness/issues/295 + +In order to use the API, you must have a valid TextSynth account and +enough credits. + +Example usage: + + python main.py --model textsynth --model_args engine=gptj_6B --no_cache --tasks piqa + +Homepage: https://textsynth.com/index.html +""" + +import logging +import os + +import requests as _requests +from tqdm import tqdm + +from lm_eval.api.model import LM +from lm_eval.api.registry import register_model +from lm_eval.models.utils import retry_on_specific_exceptions + + +logger = logging.getLogger(__name__) + + +def textsynth_completion(**kwargs): + """Query TextSynth API for completion. + Retry with back-off until they respond. + """ + + def _exception_callback(e: Exception, sleep_time: float) -> None: + import traceback + + traceback.print_exc() + + @retry_on_specific_exceptions( + on_exceptions=[_requests.exceptions.RequestException], + max_retries=None, # retry forever, consider changing + on_exception_callback=_exception_callback, + ) + def completion(): + return _requests.post(**kwargs) + + return completion() + + +@register_model("textsynth") +class TextSynthLM(LM): + def __init__(self, engine, truncate: bool = False, **kwargs) -> None: + """ + :param engine: str + TextSynth API engine (e.g. `gptj_6B`) + :param truncate: bool + Truncate input if too long (if False and input is too long, throw error) + """ + super().__init__() + + self.engine = engine + self.truncate = truncate + self.api_url = "https://api.textsynth.com" + # Read from environment variable TEXTSYNTH_API_SECRET_KEY + self.api_key = os.environ["TEXTSYNTH_API_SECRET_KEY"] + + @property + def eot_token_id(self): + # Isn't used because we override loglikelihood, loglikelihood_rolling and generate_until + raise NotImplementedError() + + @property + def max_length(self) -> int: + # NOTE: Turn on truncation to avoid errors on long inputs. + return 2048 + + @property + def max_gen_toks(self) -> int: + return 256 + + @property + def batch_size(self): + # Isn't used because we override loglikelihood, loglikelihood_rolling and generate_until + raise NotImplementedError() + + @property + def device(self): + # Isn't used because we override loglikelihood, loglikelihood_rolling and generate_until + raise NotImplementedError() + + def tok_encode(self, string: str): + # Isn't used because we override loglikelihood, loglikelihood_rolling and generate_until + raise NotImplementedError() + + def tok_decode(self, tokens): + # Isn't used because we override loglikelihood, loglikelihood_rolling and generate_until + raise NotImplementedError() + + def loglikelihood(self, requests, disable_tqdm: bool = False): + res = [] + for context, continuation in tqdm(requests, disable=disable_tqdm): + response = textsynth_completion( + url=self.api_url + "/v1/engines/" + self.engine + "/logprob", + headers={"Authorization": "Bearer " + self.api_key}, + json={"context": context, "continuation": continuation}, + ) + resp = response.json() + if "logprob" in resp: + logprob = resp["logprob"] + is_greedy = resp["is_greedy"] + res.append((logprob, is_greedy)) + + self.cache_hook.add_partial( + "loglikelihood", (context, continuation), (logprob, is_greedy) + ) + else: + logger.error( + f"The following response does not contain `logprobs`. Got:\n{resp}" + ) + assert False + return res + + def loglikelihood_rolling(self, requests, disable_tqdm: bool = False): + # TODO: The TextSynth API does not support tokenized inputs so we cannot + # manually partition long contexts into smaller rolling windows as + # done for other models derived from `BaseLM`. Override this method + # with a windowing scheme that works for direct string inputs. + raise NotImplementedError( + "`loglikelihood_rolling` is currently not supported due to lack of " + "input tokenization support from TextSynth." + ) + + def generate_until(self, requests, disable_tqdm: bool = False): + if not requests: + return [] + + res = [] + for request in tqdm(requests, disable=disable_tqdm): + inp = request[0] + request_args = request[1] + until = request_args["until"] + response = textsynth_completion( + url=self.api_url + "/v1/engines/" + self.engine + "/completions", + headers={"Authorization": "Bearer " + self.api_key}, + json={ + "prompt": inp, + "max_tokens": self.max_gen_toks, + "top_k": 1, + "stop": until, + }, + ) + resp = response.json() + if "text" in resp: + s = resp["text"] + res.append(s) + + self.cache_hook.add_partial("generate_until", (inp, request_args), s) + else: + logger.error( + "The following response does not contain generated `text`. " + "Got:\n{resp}" + ) + assert False + return res + + def _model_call(self, inps): + # Isn't used because we override _loglikelihood_tokens + raise NotImplementedError() + + def _model_generate(self, context, max_length, eos_token_id): + # Isn't used because we override generate_until + raise NotImplementedError() diff --git a/lm-evaluation-harness/lm_eval/prompts/__pycache__/__init__.cpython-311.pyc b/lm-evaluation-harness/lm_eval/prompts/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7a2c9b94e69bd6be12bf7294c83cf5a99e0dfe87 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/prompts/__pycache__/__init__.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/tasks/__pycache__/__init__.cpython-311.pyc b/lm-evaluation-harness/lm_eval/tasks/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..89e84482393ad14b78df52fd4ab42c4760358fc0 Binary files /dev/null and b/lm-evaluation-harness/lm_eval/tasks/__pycache__/__init__.cpython-311.pyc differ diff --git a/lm-evaluation-harness/lm_eval/tasks/aclue/_aclue.yaml b/lm-evaluation-harness/lm_eval/tasks/aclue/_aclue.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a2ae37ef5a6794a0db58005ea14f3943e56c87e3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aclue/_aclue.yaml @@ -0,0 +1,26 @@ +group: aclue +task: + - aclue_ancient_chinese_culture + - aclue_ancient_literature + - aclue_ancient_medical + - aclue_ancient_phonetics + - aclue_basic_ancient_chinese + - aclue_couplet_prediction + - aclue_homographic_character_resolution + - aclue_named_entity_recognition + - aclue_poetry_appreciate + - aclue_poetry_context_prediction + - aclue_poetry_quality_assessment + - aclue_poetry_sentiment_analysis + - aclue_polysemy_resolution + - aclue_reading_comprehension + - aclue_sentence_segmentation +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true + - metric: acc_norm + aggregation: mean + weight_by_size: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aclue/_default_template_yaml b/lm-evaluation-harness/lm_eval/tasks/aclue/_default_template_yaml new file mode 100644 index 0000000000000000000000000000000000000000..9505197a72cef39b25bd5ef39d65c13bd97a89ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aclue/_default_template_yaml @@ -0,0 +1,18 @@ +dataset_path: tyouisen/aclue +test_split: test +fewshot_split: dev +fewshot_config: + sampler: first_n +output_type: multiple_choice +doc_to_text: "{{Question.strip()}}\nA. {{A}}\nB. {{B}}\nC. {{C}}\nD. {{D}}\n答案:" +doc_to_choice: ["A", "B", "C", "D"] +doc_to_target: "{{['A', 'B', 'C', 'D'].index(Answer)}}" +metric_list: + - metric: acc + aggregation: mean + higher_is_better: true + - metric: acc_norm + aggregation: mean + higher_is_better: true +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_basic_ancient_chinese.yaml b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_basic_ancient_chinese.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5afb88be88b8778fde06cff3a2084bce14397174 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_basic_ancient_chinese.yaml @@ -0,0 +1,4 @@ +"dataset_name": "basic_ancient_chinese" +"description": "以下是关于古汉语知识的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "aclue_basic_ancient_chinese" diff --git a/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_poetry_appreciate.yaml b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_poetry_appreciate.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4642992674a1f159fe101859dead4509df6c8166 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_poetry_appreciate.yaml @@ -0,0 +1,4 @@ +"dataset_name": "poetry_appreciate" +"description": "以下是关于古诗词曲鉴赏的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "aclue_poetry_appreciate" diff --git a/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_polysemy_resolution.yaml b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_polysemy_resolution.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee0deea16f6bcb6906fd68e2e65bf72ea276e74a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/aclue/aclue_polysemy_resolution.yaml @@ -0,0 +1,4 @@ +"dataset_name": "polysemy_resolution" +"description": "以下是关于古文单字多义的单项选择题,请直接给出正确答案的选项。\n\n" +"include": "_default_template_yaml" +"task": "aclue_polysemy_resolution" diff --git a/lm-evaluation-harness/lm_eval/tasks/acpbench/gen_2shot/reach.yaml b/lm-evaluation-harness/lm_eval/tasks/acpbench/gen_2shot/reach.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c3a192fcc610a12093876275e3b5702c5aadbc36 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/acpbench/gen_2shot/reach.yaml @@ -0,0 +1,19 @@ +task: acp_reach_gen +dataset_name: acp_reach_gen +include: _gen_yaml_2shot +fewshot_config: + sampler: first_n + samples: + - context: "A robot is in a grid and can only move to places that are connected to its current position. The grid size is 5x5, and the locations are of the form fi-jf (e.g., f3-2f or f0-1f). The grid cells are connected to their neighbors (e.g., f1-2f is connected to the four neighbors f0-2f, f2-2f, f1-1f, and f1-3f). Some positions on the grid are locked and can be opened with a key of a matching shape. The robot has an arm that can pick up a key when the key is in same location as the robot and the arm is empty. There are 2 keys in 0 different shapes: Key key0-1 is of shape shape0, Key key0-0 is of shape shape0. Currently, the robot is at position f1-2f and its arm is empty. All the positions are open except the following: f4-2f has shape0 shaped lock. Key key0-0 is at position f1-0f. Key key0-1 is at position f1-3f. The available propositions are: (at ?r ?x) - Key ?r is at ?x location, (at-robot ?x) - Robot is at ?x location, (locked ?x) - Location ?x is locked, (holding ?k) - Robot is holding ?k, (open ?x) - Location ?x is open, and (arm-empty) - Robot's arm is empty." + question: "What proposition can never hold in any potentially reachable state?" + answer: "(locked f3-1f)" + - context: "There are several cities, each containing several locations, some of which are airports. There are also trucks, which can drive within a single city, and airplanes, which can fly between airports. The goal is to get some packages from various locations to various new locations. There are 2 trucks and 1 airplane, as well as 4 packages. There are 4 locations across 2 cities. The locations are in cities as follows: l0-0 and l0-1 are in c0; l1-0 and l1-1 are in c1. Currently, a0, p2, and t1 are at l1-0, p3 and p0 are at l0-0, t0 is at l0-1, p1 is in t1. The available propositions are: (at ?obj ?loc) - ?obj is at ?loc and (in ?obj1 ?obj2) - ?obj1 is in ?obj2." + question: "What proposition can never hold in any potentially reachable state?" + answer: "(at t0 l1-1)" +doc_to_text: "**Question**: {{context}} {{question}} Provide one proposition or None. **Final Answer**:" +filter_list: + - name: "acp_grammar_parse" + filter: + - function: "ACP_grammar_filter" + grammar_task: "act" + - function: "take_first" diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd0436e7e8ad0a80e32554fb9991158fe66df7be --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b73863adafc318c81c058efdfb3fd251cb987a95 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "For mathematical questions provided in Wolof language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1b77d56f2115fc188b5d973d5aaeb33b506830b5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "For mathematical questions provided in isiXhosa language. Supply the\ + \ accurate step by step answer to the provided question. \n\nQuestion: {{question}}\ + \ \nStep by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yaml new file mode 100644 index 0000000000000000000000000000000000000000..de15089149d133ac1f84ee89d1a20634286ed10c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yaml @@ -0,0 +1,36 @@ +tag: + - afrimgsm_cot_tasks + - afrimgsm_cot_tasks_prompt_5 +dataset_path: masakhane/afrimgsm +dataset_name: null # Overridden by language-specific config. +output_type: generate_until +training_split: train +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> + - <|eot_id|> +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +filter_list: + - name: "strict-match" + filter: + - function: "regex" + regex_pattern: "The answer is (\\-?[0-9\\.\\,]+)" + - function: "take_first" + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9032313ad485bd0d9e2b90854bb626b432dc1a46 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/direct_cot/prompt_5/afrimgsm_cot_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "For mathematical questions provided in Yoruba language. Supply the accurate\ + \ step by step answer to the provided question. \n\nQuestion: {{question}} \nStep\ + \ by step answer: " +include: afrimgsm_cot_yaml +task: afrimgsm_cot_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..768bcab970c2b6af4edbdaa0897ea4f24e1d0eb5 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5333b163698e15aa1b2548395dfa1069e188b6b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ae231d6da0ca0de8571eaaf38cb62121fe57d125 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..55e1992799923fc30ffea628a64b1d4bcab31bf0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2f8826ab4d61ab2a2803c73fd770db538a4db0f0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3aede319a2ad47e5e0e0a65c18ca8a85da812971 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c8e23103ee7a482a1f84002c3dc40bc10758e659 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4b97922fc5d1d051a0a98623a994b695dbf68c86 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1abdd50bced23bc6de7768b9f2db931c7f35b4ad --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..614510895a32f30082ccd6eb5cbdfa87766c4473 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: '{% if answer is not none %}{{question+"\nAnswer:"}}{% else %}{{"Question: "+question+"\nAnswer:"}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d3927ba8e0369a5220df6d72e4bb474b7e8af7ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a57260d94ac8b83f7371baecab3e822868dc671b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_1/afrimgsm_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49b559be5c1f62ab67498131d29ac0e0091f622d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86d8dbbca635d347640c699db7bde0cb6d950722 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..53078341b5822305b3cc5de3652f0e582313aff6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..88ae24a26d652729a5c52696f8c34fcec358dfd8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7e2ffcc32241ed36d042591921737f9c91deabcc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..137ccbcd322421d17c4ea369c34f1beac05ce597 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5bd7e53ca6ec65492978eb7e2bc95a0569b496e8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5134b3c423c88832393b6eb7b74338b8f7e97807 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f6135d99ce5181855cb78f1a0beba3f357c7082c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..db00be881c3170721857cfb3fb685b3adfec4dfc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3be8dd64c7e7aaacebb296c8f51397d4c50fe3ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..63766339e6e0cb5bbb3d9f45d1010d00de0aafd4 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yaml @@ -0,0 +1,34 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: "Give direct numerical answers for the question provided. \n\nQuestion: {{question}} \nAnswer: " +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +should_decontaminate: true +doc_to_decontamination_query: "Answer: " +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..01a54e15eea260f6a4ae5fa28e0ea4c627dc904c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5f7e74df5ba87e25f35e0e13762cd2116797fe65 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_2/afrimgsm_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04a14a1bf6b69e10873e181d46fcd51e2e901126 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3cda09e47c41ecc23f2e15b16c869fd6e3f13d87 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..49c95be2cd94d718bd6cb1754eee6a79ae6f5ae8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9d16ac8faf84785902baabb92a1a48bcf383a35a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2bbb66ff41f6c55799d77a19e29bfd8e01f0d61d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..488061a306ebcbc4e9f53e78214e25d5391475fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..928ba457061168ae1524db65a4f03f6ec202349d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bdc0c80788f1e0e0e4db9bc38d77aab7e2f0f8df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..04ec7565c10592c8ff143d0d2a944986e2870c1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..22ab7bde213b00599cee3e97ef8e4995f2ded97a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..617340d07bdab98b185c8e441e1a2e08bca3930b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..337ad6e470631085a735ee2133211c8e9258600e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..eb13aba534c504473735b6cb67c90efab5db9095 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f759e6aa6aebd9f6c6f82087dc2756d68a71025e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..50ab80df1d0589e592b072f71628395fe118ef8b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..544fa0ccc13fbbeb9eebc4f0eb2cb78b5c68e183 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yaml @@ -0,0 +1,32 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +doc_to_text: "Solve the following math question \n\nQuestion: {{question}} \nAnswer: " +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ee1c7917f6356e01b54ad4c545d3f08b2dfcf8a6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f3e21704f20ceba6e34273f8a2d5b1d87b648dda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_3/afrimgsm_translate_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..60387afce4fa5801492c968fc97a1519fdbeb5a8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7633bc3efd95541fd6f30aaa8d469fa993a0375e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ewe.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8e16ea929f8d5639f388c339671f1253554fd1d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_fra.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9828205094c1b447ce422da748ba347d448f3b0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_hau.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b8acf8d0fb40630e1c19e54ed5b87687f2bf2897 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_ibo.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74ac117344b5625f1a8671cb938e7d911e849eea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cf113619e461a91f0a4517dae08109055ed743f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5dffbdb899140911f0dab979581fc7b0d52674bf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..30f776f4b6c6a05a98b79234dbf45f269170dc0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_orm.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..63efa2505e80032a7ef347f1f986f0ec0952c07c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..19b86220a5eeb0319c7732add1ed7d969ab72c67 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_sot.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sot +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..20236ae8e25cbcc8cd43dd54157614518eea26d8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5fe7e7475ae49642169275f3aff639ba4d3fbeda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_twi.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a8fb5640def2bb261c461a80b66ba50324231317 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_wol.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fdb63749c47b38edfb97d20998f4fb8d4479075d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d3903948dd4a8ae2c285e33eb2551f554d2310c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yaml @@ -0,0 +1,31 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5cb74d41b85d71f851f6c3f160edbc85d006151 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_yor.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f0a068e9f662d567bb01a208e46a6e0b2d014d2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_4/afrimgsm_translate_zul.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: zul +doc_to_text: "Answer the given question with the appropriate numerical value, ensuring\ + \ that the response is clear and without any supplementary information. \n\nQuestion:\ + \ {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..48ca09aaafafe7360ca9b0c2872d99e63ff16619 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_amh.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: amh +doc_to_text: "For mathematical questions provided in Amharic language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a4a254f0a2d766380528dde5d2ad9ac7eb67bb2d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ewe.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ewe +doc_to_text: "For mathematical questions provided in Ewe language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ac62304550bb0f5b8705c7a9ba3f5934278be55d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_fra.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: fra +doc_to_text: "For mathematical questions provided in French language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..695f1f373464fb93811f28644c0b849fe72de9ed --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_hau.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: hau +doc_to_text: "For mathematical questions provided in Hausa language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7fd530e7409124357c091d42cbaf5608473976c0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_ibo.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: ibo +doc_to_text: "For mathematical questions provided in Igbo language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52ea0a78a2053f7958583167d613ca209b43e22a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_kin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: kin +doc_to_text: "For mathematical questions provided in Kinyarwanda language. Supply\ + \ the accurate numeric answer to the provided question. \n\nQuestion: {{question}}\ + \ \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_kin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..07cf6a6b0e871bbf004d809d4fcff2f7f063a2a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lin.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lin +doc_to_text: "For mathematical questions provided in Lingala language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fa3461beb8fc3c9f05dd1a58c5ca7e7de4ac6cb3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_lug.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: lug +doc_to_text: "For mathematical questions provided in Luganda language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_lug_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1a00385f578feeb5fb5071c7f2835574a68a30e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_orm.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: orm +doc_to_text: "For mathematical questions provided in Oromo language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_orm_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7f08a786ecac0379aa660a26e5619055e97063a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_sna.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: sna +doc_to_text: "For mathematical questions provided in chiShona language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_sna_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a950c84d3c617353fd9492e6a1d2a028fd836881 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_swa.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: swa +doc_to_text: "For mathematical questions provided in Swahili language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a0488295249513408b1d33de3a246b663cc523a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_twi.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: twi +doc_to_text: "For mathematical questions provided in Twi language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..61ffc3f938f8cf13da5bac35bed1c6d2a9323acf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_wol.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: wol +doc_to_text: "For mathematical questions provided in Wolof language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c308cc7f57b90ddfe959132e75aad3cc5a0b6f01 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_xho.yaml @@ -0,0 +1,7 @@ +# Generated by utils.py +dataset_name: xho +doc_to_text: "For mathematical questions provided in isiXhosa language. Supply the\ + \ accurate numeric answer to the provided question. \n\nQuestion: {{question}} \n\ + Answer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_xho_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d3903948dd4a8ae2c285e33eb2551f554d2310c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yaml @@ -0,0 +1,31 @@ +tag: afrimgsm_tt_tasks +dataset_path: masakhane/afrimgsm-translate-test +output_type: generate_until +test_split: test +doc_to_target: '{% if answer is not none %}{{answer[21:]}}{% else %}{{answer_number|string}}{% endif %}' +target_delimiter: "" +generation_kwargs: + do_sample: false + until: + - 'Question:' + - + - <|im_end|> +filter_list: + - name: remove_whitespace + filter: + - function: remove_whitespace + - function: take_first + - filter: + - function: regex + group_select: -1 + regex_pattern: (-?[$0-9.,]{2,})|(-?[0-9]+) + - function: take_first + name: flexible-extract +metric_list: + - metric: exact_match + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true +metadata: + version: 2.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f2a0a0fd28fc895e25d309a8f5aaf64769e7de6c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate/prompt_5/afrimgsm_translate_yor.yaml @@ -0,0 +1,6 @@ +# Generated by utils.py +dataset_name: yor +doc_to_text: "For mathematical questions provided in Yoruba language. Supply the accurate\ + \ numeric answer to the provided question. \n\nQuestion: {{question}} \nAnswer: " +include: afrimgsm_translate_yaml +task: afrimgsm_translate_yor_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d43ddd233b785cbfba006785de6db94bb4eb5d97 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/afrimgsm_tt_cot.yaml @@ -0,0 +1,9 @@ +group: afrimgsm_tt_cot-irokobench +task: + - afrimgsm_tt_cot_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..da7764e81c0665c53c129f42d61629460a74ea1e --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a16b91ffb250880c9217e63d6e8c1e46c7d4021c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3bee8575de4f774c0ae7510e3a074023919dbbe3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..400bf8887718fbe40d6701cb2121a3cd271c1360 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_kin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83c9565d54a2786fea141b6775681172cf49b592 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca19eb14ef29d16affc7efe255e3faa3ae4deb06 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_orm_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9f8fc2ef28d888800670a9f3068f0d58d670d5ec --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_sot_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0545cccda86167036dc321b11d29c1ca1ca2542 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6f340a46529bd305f4bdc5ea73b7cd148ec6d7d1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cf093766dc687b5a092d984ab8cda6514ffd56a9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_1/afrimgsm_cot_translate_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e9f97735373ee9e47f6234b27002f6f12edb1a13 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_amh_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1a83764758c3e2540d3a097ed0e0f5ac987604a1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..b496c775817645ce477d84749de8e71f0badbc22 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_fra_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1022ae899b0e2413351e01aafef9de08b00688ba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..dd2a2528ef1c104f1714735f1a5b753c10966607 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a774c189513e159a0d9cdf034cd0470cc25d8b84 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimgsm/translate_cot/prompt_2/afrimgsm_cot_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimgsm_cot_translate_yaml +task: afrimgsm_cot_translate_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e85bd7dc7dcc4fa2f5ea90f4f540a0a75b160dbd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3301647298d216de664ff07e2c8a10e134afe388 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimmlu_direct +task: afrimmlu_direct_lug_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..17222f95253270fdcff74177fbd0474cb75660b3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimmlu_direct +task: afrimmlu_direct_sna_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c5ebed9f96e857356adf2ccaa2de2cf818874e71 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimmlu_direct +task: afrimmlu_direct_swa_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..fb270c949a7a250e4f4810a5086a80ff22716f1f --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4ccbc47cd02c5de0c886deed9f4549141884eac8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3e30d2017740585b5c675ec898fa9d9512e4ac52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimmlu_direct +task: afrimmlu_direct_xho_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e3de56f8d3c9fe1f283712ea359ead734a413d93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..86c56fec097dc4c636070f6c0ab0750a23bb2435 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/afrimmlu_direct_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimmlu_direct +task: afrimmlu_direct_zul_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..f1bb9162f0fbc68807db68134970ae2636980cbf --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_1/utils.py @@ -0,0 +1,32 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_choice(doc): + choices = eval(doc["choices"]) + return choices + + +def doc_to_text(doc): + output = """You are a highly knowledgeable and intelligent artificial intelligence + model answers multiple-choice questions about {subject} + + Question: {question} + + Choices: + A: {choice1} + B: {choice2} + C: {choice3} + D: {choice4} + + Answer: """ + + choices = eval(doc["choices"]) + text = output.format( + subject=doc["subject"], + question=doc["question"], + choice1=choices[0], + choice2=choices[1], + choice3=choices[2], + choice4=choices[3], + ) + return text diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..fefabf7e0b52e644d1e9d922c8f899607eab6075 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_2 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..26acfcfa93b3019a746a5e9a78e4cdb48871c978 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..29b4a4d2029e945c4bf52654a819dcb89b898431 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0cf7db0e4c0585c2cde8f0645e64438546ba5818 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ce7c2e896509b874459deb156f2ba34287a908c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimmlu_direct +task: afrimmlu_direct_kin_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f4c57ae36fbad9a75737562b6cb619b53e933c34 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimmlu_direct +task: afrimmlu_direct_lug_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..494d4240693fe9907f965cd8ad5ccc71fcfc2868 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimmlu_direct +task: afrimmlu_direct_orm_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7706ad64ccc0baef8fc4964e61871805187db548 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimmlu_direct +task: afrimmlu_direct_sna_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..353bd2574657f0b7f49d0be77edd777893ac549b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimmlu_direct +task: afrimmlu_direct_sot_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..54a16c6c2f5839c6aec545c74f6a5df3293938df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimmlu_direct +task: afrimmlu_direct_swa_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8bb35bd5f9ad784fbf2101e3ce14e82710c75858 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..963f7cd2cdee47fda381af3cbe1a63c56601b709 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9da0589a8bfe5791ff3764fd26e1f375a836b701 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimmlu_direct +task: afrimmlu_direct_xho_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..39b365418eda1be6e4c344d4e32717045d2eafda --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8766392a0dc7e8abd5b8145f37bcd13f424a8b6a --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/afrimmlu_direct_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimmlu_direct +task: afrimmlu_direct_zul_prompt_2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..e0cfb334c27cfe4c5bbb1ff7126215c0ea9130c9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_2/utils.py @@ -0,0 +1,30 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_choice(doc): + choices = eval(doc["choices"]) + return choices + + +def doc_to_text(doc): + output = """As an expert in {subject}, choose the most accurate answer to the question below. +Your goal is to select the correct option 'A', 'B', 'C', or 'D' by understanding the nuances of the topic. + +Question: {question} +Choices: + A: {choice1} + B: {choice2} + C: {choice3} + D: {choice4} +Answer: """ + + choices = eval(doc["choices"]) + text = output.format( + subject=doc["subject"], + question=doc["question"], + choice1=choices[0], + choice2=choices[1], + choice3=choices[2], + choice4=choices[3], + ) + return text diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..fb2fd165fcba0457c82e825afa5d8252546dc09c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_3 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c7c28f20b09193f8a0a5c1c0f4ffd8ae59312a08 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimmlu_direct +task: afrimmlu_direct_amh_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83f7cfcb32c1d85061a3d9b6e1cca169a61d4ff0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..351bdf330c4b30d85448e45b9233aaf6cb704c4b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..691978187578805d95bc215c8d678273f02d343d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimmlu_direct +task: afrimmlu_direct_fra_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..90521523bef4afe8420fc579108dd2353afb49f3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..43a88fe6c13d804531aa7e251fe86c0102562bc9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..977f3ab259efac5e6afcccdfd44e04279651ad18 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimmlu_direct +task: afrimmlu_direct_kin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2d25584a3fe0d9c7a73e13ab7a3f1b8652616efd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2b4da1a7f550f66c1b3f084879413d2d9fc13641 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimmlu_direct +task: afrimmlu_direct_lug_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2738f980d41cea65fa790aa16232a0f6a7584226 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimmlu_direct +task: afrimmlu_direct_orm_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sna.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sna.yaml new file mode 100644 index 0000000000000000000000000000000000000000..063d111ac4aa0c8625d8615e7b13c1d10ac906fb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sna.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sna +include: afrimmlu_direct +task: afrimmlu_direct_sna_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6cf6e66d4fb91e3e98ed5b0fe955379bcb31bf26 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimmlu_direct +task: afrimmlu_direct_sot_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e90204d40f2fef22b46b190b552cd9f9fcb777b0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimmlu_direct +task: afrimmlu_direct_swa_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..719ebe9002cc9fab19dfa793153d779ad8ffbee6 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f0f1d0d709b8f0dddd5708b66f6d27a122984d0 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8fc1af4d171021226243c69a59b44363b4a16639 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimmlu_direct +task: afrimmlu_direct_xho_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..a641b03ae173f2342e8d7178119e68fea2e5f000 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8c6b493d34a99a4250676c2f0130ceff6b4ea4f8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/afrimmlu_direct_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimmlu_direct +task: afrimmlu_direct_zul_prompt_3 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/utils.py b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/utils.py new file mode 100644 index 0000000000000000000000000000000000000000..bc3da2e29667b4b25f68757e2169a5c8aa0c8dea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_3/utils.py @@ -0,0 +1,32 @@ +from lm_eval.utils import weighted_f1_score + + +def doc_to_choice(doc): + choices = eval(doc["choices"]) + return choices + + +def doc_to_text(doc): + output = """You are a subject matter expert in {subject}. + + Utilizing your expertise in {subject}, answer the following multiple-choice question + by picking 'A', 'B', 'C', or 'D'. + +Question: {question} +Choices: + A: {choice1} + B: {choice2} + C: {choice3} + D: {choice4} +Answer: """ + + choices = eval(doc["choices"]) + text = output.format( + subject=doc["subject"], + question=doc["question"], + choice1=choices[0], + choice2=choices[1], + choice3=choices[2], + choice4=choices[3], + ) + return text diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..c15b7b2fc3991517b15f2c370a246adb907f2e52 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_4 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cc862dc2327d23f394742ee51031e37fc7c97ff1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimmlu_direct +task: afrimmlu_direct_amh_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..69baef502b78bce307b77a7c44c8c4323ebc1102 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f5af1074f4b993daa4f2468f60baf83b48bfc470 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d1f94eea1e47bd9442dbc76181069bf47ac29da1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimmlu_direct +task: afrimmlu_direct_fra_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ca8f7c5ed0ae15bc5a5e96c776f2251f2cff06fa --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8a181d07cc6a81aaf42308fc137c2241a0d8d444 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_kin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_kin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..8f86122466a7db1fdc06a2352385fdc1fc78bd69 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_kin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: kin +include: afrimmlu_direct +task: afrimmlu_direct_kin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..3c7d3ecf7a86a68248a49d5fc97947ca8da69b0b --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..467201319f12257549de2fa3c260591dee13f311 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimmlu_direct +task: afrimmlu_direct_lug_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_orm.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_orm.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e52668253d495c2583d9f5e964dc73c6850a5729 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_orm.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: orm +include: afrimmlu_direct +task: afrimmlu_direct_orm_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_sot.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_sot.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0342dc10b71f69880b3a3352f9d9def10f9815c8 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_sot.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: sot +include: afrimmlu_direct +task: afrimmlu_direct_sot_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ec9a3525f534ac5ea7ed6817d8f52bc67b57c445 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimmlu_direct +task: afrimmlu_direct_swa_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..83dc916c68120a6a28234dfaad48b71c9cbfdba3 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e656af2c6697004f7ea94ff638b8b8d4f9f8d549 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_xho.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_xho.yaml new file mode 100644 index 0000000000000000000000000000000000000000..ab23d9346400d5ec0eedb7f91f530a04499f163c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_xho.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: xho +include: afrimmlu_direct +task: afrimmlu_direct_xho_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_yor.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_yor.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0dd0254819a9ef2b6b2b795bbcfad7ed8ef1c314 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_yor.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: yor +include: afrimmlu_direct +task: afrimmlu_direct_yor_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_zul.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_zul.yaml new file mode 100644 index 0000000000000000000000000000000000000000..98a0937fc74a67a0fae7faa39164f10c288aa3f7 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_4/afrimmlu_direct_zul.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: zul +include: afrimmlu_direct +task: afrimmlu_direct_zul_prompt_4 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct new file mode 100644 index 0000000000000000000000000000000000000000..3da1eb827af65c9bcb69dd4af7eab06df848ade2 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct @@ -0,0 +1,37 @@ +tag: + - afrimmlu_tasks + - afrimmlu_tasks_prompt_5 + - afrobench_mmlu_tasks +dataset_path: masakhane/afrimmlu +dataset_name: null +output_type: multiple_choice +validation_split: validation +test_split: test +fewshot_split: validation +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_amh.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_amh.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cff031d7936a5b82bacf175d8d17e54a51d7fe92 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_amh.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: amh +include: afrimmlu_direct +task: afrimmlu_direct_amh_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_eng.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_eng.yaml new file mode 100644 index 0000000000000000000000000000000000000000..52f317982b39dd21a3e11ce6695b95af9ef8f8df --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_eng.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: eng +include: afrimmlu_direct +task: afrimmlu_direct_eng_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cef2f86599b589c38e7fb30723a3024b3fcfeffe --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_direct +task: afrimmlu_direct_ewe_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_fra.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_fra.yaml new file mode 100644 index 0000000000000000000000000000000000000000..042c0bbbf0c5afc60776127770b88d15c7c7c5ea --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_fra.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: fra +include: afrimmlu_direct +task: afrimmlu_direct_fra_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cd507182558a8884859713d4fbcf356898d9176c --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_direct +task: afrimmlu_direct_hau_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..9e9839001ed95c56ae13b1fa97466fbbb39c4acb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_direct +task: afrimmlu_direct_ibo_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6eca1f8e7ce1b442be6b62ec924143d736f97c62 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_direct +task: afrimmlu_direct_lin_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_swa.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_swa.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c1cd2672b09792c6fd0cf6a9c3c77b83ac5cdbcb --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_swa.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: swa +include: afrimmlu_direct +task: afrimmlu_direct_swa_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_twi.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_twi.yaml new file mode 100644 index 0000000000000000000000000000000000000000..1e2e258c6d020e6599fe3d8fd92388958fce14b1 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_twi.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: twi +include: afrimmlu_direct +task: afrimmlu_direct_twi_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_wol.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_wol.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d721871b35a08b564c04af09ca32349a4433bf93 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/direct/prompt_5/afrimmlu_direct_wol.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: wol +include: afrimmlu_direct +task: afrimmlu_direct_wol_prompt_5 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/afrimmlu_tt.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/afrimmlu_tt.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bbbf9387e5563914e7b89a06540d73156c8fb1b9 --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/afrimmlu_tt.yaml @@ -0,0 +1,9 @@ +group: afrimmlu_tt-irokobench +task: + - afrimmlu_tt_tasks +aggregate_metric_list: + - metric: acc + aggregation: mean + weight_by_size: true +metadata: + version: 2 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate new file mode 100644 index 0000000000000000000000000000000000000000..7a974279a3918de90369c391b09de818cb1b483d --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate @@ -0,0 +1,32 @@ +tag: afrimmlu_tt_tasks +dataset_path: masakhane/afrimmlu-translate-test +dataset_name: null +output_type: multiple_choice +test_split: test +doc_to_text: !function utils.doc_to_text +doc_to_target: "{{['A', 'B', 'C', 'D'].index(answer)}}" +doc_to_choice: !function utils.doc_to_choice +should_decontaminate: true +doc_to_decontamination_query: "Question: {{question}}\nAnswer:" +metric_list: + - metric: f1 + aggregation: !function utils.weighted_f1_score + # aggregation: mean + average: weighted + hf_evaluate: true + higher_is_better: True + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" + - metric: acc + aggregation: mean + higher_is_better: true + ignore_case: true + ignore_punctuation: true + regexes_to_ignore: + - "," + - "\\$" +metadata: + version: 1.0 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ewe.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ewe.yaml new file mode 100644 index 0000000000000000000000000000000000000000..45298a1f4ed4c0f101f816b7adc46115da2b3aba --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ewe.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ewe +include: afrimmlu_translate +task: afrimmlu_translate_ewe_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_hau.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_hau.yaml new file mode 100644 index 0000000000000000000000000000000000000000..c09424d3812e26e2c8ff8a2bc08eecec7690f9fd --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_hau.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: hau +include: afrimmlu_translate +task: afrimmlu_translate_hau_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ibo.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ibo.yaml new file mode 100644 index 0000000000000000000000000000000000000000..6fe139910f24c12612a64c9633fd2dd625c580cc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_ibo.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: ibo +include: afrimmlu_translate +task: afrimmlu_translate_ibo_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lin.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lin.yaml new file mode 100644 index 0000000000000000000000000000000000000000..e245d7bdd3b96e16b84da7db505d0b005cb165fc --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lin.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lin +include: afrimmlu_translate +task: afrimmlu_translate_lin_prompt_1 diff --git a/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lug.yaml b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lug.yaml new file mode 100644 index 0000000000000000000000000000000000000000..bcbac5f65455134c60b873710d7c2e66cf0ac7ca --- /dev/null +++ b/lm-evaluation-harness/lm_eval/tasks/afrimmlu/translate/prompt_1/afrimmlu_translate_lug.yaml @@ -0,0 +1,4 @@ +# Generated by utils.py +dataset_name: lug +include: afrimmlu_translate +task: afrimmlu_translate_lug_prompt_1