Hummingbird-V2 / evaluation /open_slm.json
juinron's picture
Publish Hummingbird-V2 10B base model and model card
ebd2f40 verified
Raw
History Blame Contribute Delete
10.9 kB
{
"config": {
"batch_size": 64,
"batch_sizes": [],
"bootstrap_iters": 100000,
"device": "cuda",
"fewshot_seed": 1234,
"gen_kwargs": null,
"limit": null,
"model": "MicroLoopHarnessLM",
"model_args": null,
"numpy_seed": 1234,
"random_seed": 0,
"torch_seed": 1234,
"use_cache": null
},
"configs": {
"arc_challenge": {
"dataset_name": "ARC-Challenge",
"dataset_path": "allenai/ai2_arc",
"description": "",
"doc_to_choice": "{{choices.text}}",
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"doc_to_text": "Question: {{question}}\nAnswer:",
"fewshot_config": {
"doc_to_choice": "{{choices.text}}",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"doc_to_text": "Question: {{question}}\nAnswer:",
"fewshot_delimiter": "\n\n",
"fewshot_indices": null,
"gen_prefix": null,
"process_docs": null,
"sampler": "default",
"samples": null,
"split": null,
"target_delimiter": " "
},
"fewshot_delimiter": "\n\n",
"metadata": {
"config_source": "D:\\llm\\hummingbird\\.venv\\Lib\\site-packages\\lm_eval\\tasks\\arc\\arc_challenge.yaml",
"version": 1.0
},
"metric_list": [
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc"
},
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc_norm"
}
],
"num_fewshot": 0,
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"target_delimiter": " ",
"task": "arc_challenge",
"test_split": "test",
"training_split": "train",
"unsafe_code": false,
"validation_split": "validation"
},
"arc_easy": {
"dataset_name": "ARC-Easy",
"dataset_path": "allenai/ai2_arc",
"description": "",
"doc_to_choice": "{{choices.text}}",
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"doc_to_text": "Question: {{question}}\nAnswer:",
"fewshot_config": {
"doc_to_choice": "{{choices.text}}",
"doc_to_target": "{{choices.label.index(answerKey)}}",
"doc_to_text": "Question: {{question}}\nAnswer:",
"fewshot_delimiter": "\n\n",
"fewshot_indices": null,
"gen_prefix": null,
"process_docs": null,
"sampler": "default",
"samples": null,
"split": null,
"target_delimiter": " "
},
"fewshot_delimiter": "\n\n",
"metadata": {
"config_source": "D:\\llm\\hummingbird\\.venv\\Lib\\site-packages\\lm_eval\\tasks\\arc\\arc_easy.yaml",
"version": 1.0
},
"metric_list": [
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc"
},
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc_norm"
}
],
"num_fewshot": 0,
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"target_delimiter": " ",
"task": "arc_easy",
"test_split": "test",
"training_split": "train",
"unsafe_code": false,
"validation_split": "validation"
},
"hellaswag": {
"dataset_path": "Rowan/hellaswag",
"description": "",
"doc_to_choice": "choices",
"doc_to_target": "{{label}}",
"doc_to_text": "{{query}}",
"fewshot_config": {
"doc_to_choice": "choices",
"doc_to_target": "{{label}}",
"doc_to_text": "{{query}}",
"fewshot_delimiter": "\n\n",
"fewshot_indices": null,
"gen_prefix": null,
"process_docs": "<callable function>",
"sampler": "default",
"samples": null,
"split": null,
"target_delimiter": " "
},
"fewshot_delimiter": "\n\n",
"metadata": {
"config_source": "D:\\llm\\hummingbird\\.venv\\Lib\\site-packages\\lm_eval\\tasks\\hellaswag\\hellaswag.yaml",
"version": 1.0
},
"metric_list": [
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc"
},
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc_norm"
}
],
"num_fewshot": 0,
"output_type": "multiple_choice",
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
"repeats": 1,
"should_decontaminate": false,
"target_delimiter": " ",
"task": "hellaswag",
"training_split": "train",
"unsafe_code": false,
"validation_split": "validation"
},
"piqa": {
"dataset_path": "baber/piqa",
"description": "",
"doc_to_choice": "{{[sol1, sol2]}}",
"doc_to_decontamination_query": "goal",
"doc_to_target": "label",
"doc_to_text": "Question: {{goal}}\nAnswer:",
"fewshot_config": {
"doc_to_choice": "{{[sol1, sol2]}}",
"doc_to_target": "label",
"doc_to_text": "Question: {{goal}}\nAnswer:",
"fewshot_delimiter": "\n\n",
"fewshot_indices": null,
"gen_prefix": null,
"process_docs": null,
"sampler": "default",
"samples": null,
"split": null,
"target_delimiter": " "
},
"fewshot_delimiter": "\n\n",
"metadata": {
"config_source": "D:\\llm\\hummingbird\\.venv\\Lib\\site-packages\\lm_eval\\tasks\\piqa\\piqa.yaml",
"version": 1.0
},
"metric_list": [
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc"
},
{
"aggregation": "mean",
"higher_is_better": true,
"metric": "acc_norm"
}
],
"num_fewshot": 0,
"output_type": "multiple_choice",
"repeats": 1,
"should_decontaminate": true,
"target_delimiter": " ",
"task": "piqa",
"training_split": "train",
"unsafe_code": false,
"validation_split": "validation"
}
},
"date": 1790004139.4918497,
"frost_evaluation": {
"batch_size": 64,
"checkpoint": "D:\\llm\\hummingbird\\artifacts\\runs\\flightmix_10b_balanced_10m_muon_lrfix_20260921\\checkpoint-tokens-10000000000",
"device": "cuda",
"loop_count": 1,
"num_fewshot": 0,
"numeric_serialization": "none",
"qk_norm_position": null,
"tokenizer": "D:\\llm\\hummingbird\\artifacts\\runs\\e3_tokenizers\\tok_4k_digit"
},
"git_hash": "9ac6cbb",
"group_subtasks": {},
"higher_is_better": {
"arc_challenge": {
"acc": true,
"acc_norm": true
},
"arc_easy": {
"acc": true,
"acc_norm": true
},
"hellaswag": {
"acc": true,
"acc_norm": true
},
"piqa": {
"acc": true,
"acc_norm": true
}
},
"lm_eval_version": "0.4.12",
"n-samples": {
"arc_challenge": {
"effective": 1172,
"original": 1172
},
"arc_easy": {
"effective": 2376,
"original": 2376
},
"hellaswag": {
"effective": 10042,
"original": 10042
},
"piqa": {
"effective": 1838,
"original": 1838
}
},
"n-shot": {
"arc_challenge": 0,
"arc_easy": 0,
"hellaswag": 0,
"piqa": 0
},
"pretty_env_info": "PyTorch version: 2.10.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Microsoft Windows 11 Pro (10.0.26200 64-bit)\nGCC version: Could not collect\nClang version: Could not collect\nCMake version: version 4.3.3\nLibc version: N/A\n\nPython version: 3.13.2 (tags/v3.13.2:4f8bb39, Feb 4 2025, 15:23:48) [MSC v.1942 64 bit (AMD64)] (64-bit runtime)\nPython platform: Windows-11-10.0.26200-SP0\nIs CUDA available: True\nCUDA runtime version: 13.3.73\r\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: GPU 0: NVIDIA GeForce RTX 4070\nNvidia driver version: Could not collect\ncuDNN version: Could not collect\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nName: AMD Ryzen 5 5600X 6-Core Processor \nManufacturer: AuthenticAMD\nFamily: 107\nArchitecture: 9\nProcessorType: 3\nDeviceID: CPU0\nCurrentClockSpeed: 3701\nMaxClockSpeed: 3701\nL2CacheSize: 3072\nL2CacheSpeed: None\nRevision: 8450\n\nVersions of relevant libraries:\n[pip3] numpy==2.3.5\n[pip3] torch==2.10.0+cu128\n[pip3] triton-windows==3.6.0.post26\n[conda] Could not collect",
"results": {
"arc_challenge": {
"acc,none": 0.17918088737201365,
"acc_norm,none": 0.21160409556313994,
"acc_norm_stderr,none": 0.011935916358632793,
"acc_stderr,none": 0.011207045216615686,
"alias": "arc_challenge",
"name": "arc_challenge",
"sample_len": 1172
},
"arc_easy": {
"acc,none": 0.4048821548821549,
"acc_norm,none": 0.3939393939393939,
"acc_norm_stderr,none": 0.010026305355981951,
"acc_stderr,none": 0.010072423960395665,
"alias": "arc_easy",
"name": "arc_easy",
"sample_len": 2376
},
"hellaswag": {
"acc,none": 0.2689703246365266,
"acc_norm,none": 0.27633937462656843,
"acc_norm_stderr,none": 0.004462727543055927,
"acc_stderr,none": 0.004425182676353205,
"alias": "hellaswag",
"name": "hellaswag",
"sample_len": 10042
},
"piqa": {
"acc,none": 0.588683351468988,
"acc_norm,none": 0.573993471164309,
"acc_norm_stderr,none": 0.011537375448519308,
"acc_stderr,none": 0.01148086057719274,
"alias": "piqa",
"name": "piqa",
"sample_len": 1838
}
},
"transformers_version": "5.14.1",
"upper_git_hash": null,
"versions": {
"arc_challenge": 1.0,
"arc_easy": 1.0,
"hellaswag": 1.0,
"piqa": 1.0
}
}