443 lines
18 KiB
JSON
443 lines
18 KiB
JSON
{
|
|
"results": {
|
|
"arc_easy": {
|
|
"name": "arc_easy",
|
|
"alias": "arc_easy",
|
|
"sample_len": 2376,
|
|
"acc,none": 0.43897306397306396,
|
|
"acc_stderr,none": 0.01018307601297191,
|
|
"acc_norm,none": 0.3897306397306397,
|
|
"acc_norm_stderr,none": 0.010007169391797084
|
|
},
|
|
"hellaswag": {
|
|
"name": "hellaswag",
|
|
"alias": "hellaswag",
|
|
"sample_len": 10042,
|
|
"acc,none": 0.27185819557857,
|
|
"acc_stderr,none": 0.0044400791732768724,
|
|
"acc_norm,none": 0.280920135431189,
|
|
"acc_norm_stderr,none": 0.004485300194072218
|
|
},
|
|
"piqa": {
|
|
"name": "piqa",
|
|
"alias": "piqa",
|
|
"sample_len": 1838,
|
|
"acc,none": 0.5799782372143635,
|
|
"acc_stderr,none": 0.011515615810587429,
|
|
"acc_norm,none": 0.5783460282916213,
|
|
"acc_norm_stderr,none": 0.011521722161800405
|
|
},
|
|
"winogrande": {
|
|
"name": "winogrande",
|
|
"alias": "winogrande",
|
|
"sample_len": 1267,
|
|
"acc,none": 0.5327545382794001,
|
|
"acc_stderr,none": 0.014022300570434274
|
|
},
|
|
"lambada_openai": {
|
|
"name": "lambada_openai",
|
|
"alias": "lambada_openai",
|
|
"sample_len": 5153,
|
|
"perplexity,none": 932.9038478046385,
|
|
"perplexity_stderr,none": 47.08239338640588,
|
|
"acc,none": 0.11663108868620221,
|
|
"acc_stderr,none": 0.004471881565476763
|
|
},
|
|
"boolq": {
|
|
"name": "boolq",
|
|
"alias": "boolq",
|
|
"sample_len": 3270,
|
|
"acc,none": 0.5749235474006116,
|
|
"acc_stderr,none": 0.008646316159373297
|
|
}
|
|
},
|
|
"group_subtasks": {},
|
|
"configs": {
|
|
"arc_easy": {
|
|
"task": "arc_easy",
|
|
"dataset_path": "allenai/ai2_arc",
|
|
"dataset_name": "ARC-Easy",
|
|
"training_split": "train",
|
|
"validation_split": "validation",
|
|
"test_split": "test",
|
|
"doc_to_text": "Question: {{question}}\nAnswer:",
|
|
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
|
"unsafe_code": false,
|
|
"doc_to_choice": "{{choices.text}}",
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": null,
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "Question: {{question}}\nAnswer:",
|
|
"doc_to_choice": "{{choices.text}}",
|
|
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "acc",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
},
|
|
{
|
|
"metric": "acc_norm",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
}
|
|
],
|
|
"output_type": "multiple_choice",
|
|
"repeats": 1,
|
|
"should_decontaminate": true,
|
|
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
|
|
"metadata": {
|
|
"version": 1.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/arc/arc_easy.yaml"
|
|
}
|
|
},
|
|
"boolq": {
|
|
"task": "boolq",
|
|
"dataset_path": "aps/super_glue",
|
|
"dataset_name": "boolq",
|
|
"training_split": "train",
|
|
"validation_split": "validation",
|
|
"doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
|
|
"doc_to_target": "label",
|
|
"unsafe_code": false,
|
|
"doc_to_choice": [
|
|
"no",
|
|
"yes"
|
|
],
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": null,
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "{{passage}}\nQuestion: {{question}}?\nAnswer:",
|
|
"doc_to_choice": [
|
|
"no",
|
|
"yes"
|
|
],
|
|
"doc_to_target": "label",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "acc"
|
|
}
|
|
],
|
|
"output_type": "multiple_choice",
|
|
"repeats": 1,
|
|
"should_decontaminate": true,
|
|
"doc_to_decontamination_query": "passage",
|
|
"metadata": {
|
|
"version": 2.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/super_glue/boolq/default.yaml"
|
|
}
|
|
},
|
|
"hellaswag": {
|
|
"task": "hellaswag",
|
|
"dataset_path": "Rowan/hellaswag",
|
|
"training_split": "train",
|
|
"validation_split": "validation",
|
|
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
|
"doc_to_text": "{{query}}",
|
|
"doc_to_target": "{{label}}",
|
|
"unsafe_code": false,
|
|
"doc_to_choice": "choices",
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": "<function process_docs at 0x78a5a660dbc0>",
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "{{query}}",
|
|
"doc_to_choice": "choices",
|
|
"doc_to_target": "{{label}}",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "acc",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
},
|
|
{
|
|
"metric": "acc_norm",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
}
|
|
],
|
|
"output_type": "multiple_choice",
|
|
"repeats": 1,
|
|
"should_decontaminate": false,
|
|
"metadata": {
|
|
"version": 1.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/hellaswag/hellaswag.yaml"
|
|
}
|
|
},
|
|
"lambada_openai": {
|
|
"task": "lambada_openai",
|
|
"dataset_path": "EleutherAI/lambada_openai",
|
|
"dataset_name": "default",
|
|
"test_split": "test",
|
|
"doc_to_text": "{{text.split(' ')[:-1]|join(' ')}}",
|
|
"doc_to_target": "{{' '+text.split(' ')[-1]}}",
|
|
"unsafe_code": false,
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": null,
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "{{text.split(' ')[:-1]|join(' ')}}",
|
|
"doc_to_choice": null,
|
|
"doc_to_target": "{{' '+text.split(' ')[-1]}}",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "perplexity",
|
|
"aggregation": "perplexity",
|
|
"higher_is_better": false
|
|
},
|
|
{
|
|
"metric": "acc",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
}
|
|
],
|
|
"output_type": "loglikelihood",
|
|
"repeats": 1,
|
|
"should_decontaminate": true,
|
|
"doc_to_decontamination_query": "{{text}}",
|
|
"metadata": {
|
|
"version": 1.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/lambada/lambada_openai.yaml"
|
|
}
|
|
},
|
|
"piqa": {
|
|
"task": "piqa",
|
|
"dataset_path": "baber/piqa",
|
|
"training_split": "train",
|
|
"validation_split": "validation",
|
|
"doc_to_text": "Question: {{goal}}\nAnswer:",
|
|
"doc_to_target": "label",
|
|
"unsafe_code": false,
|
|
"doc_to_choice": "{{[sol1, sol2]}}",
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": null,
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "Question: {{goal}}\nAnswer:",
|
|
"doc_to_choice": "{{[sol1, sol2]}}",
|
|
"doc_to_target": "label",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "acc",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
},
|
|
{
|
|
"metric": "acc_norm",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
}
|
|
],
|
|
"output_type": "multiple_choice",
|
|
"repeats": 1,
|
|
"should_decontaminate": true,
|
|
"doc_to_decontamination_query": "goal",
|
|
"metadata": {
|
|
"version": 1.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/piqa/piqa.yaml"
|
|
}
|
|
},
|
|
"winogrande": {
|
|
"task": "winogrande",
|
|
"dataset_path": "allenai/winogrande",
|
|
"dataset_name": "winogrande_xl",
|
|
"training_split": "train",
|
|
"validation_split": "validation",
|
|
"doc_to_text": "def doc_to_text(doc):\n answer_to_num = {\"1\": 0, \"2\": 1}\n return answer_to_num[doc[\"answer\"]]\n",
|
|
"doc_to_target": "def doc_to_target(doc):\n idx = doc[\"sentence\"].index(\"_\") + 1\n return doc[\"sentence\"][idx:].strip()\n",
|
|
"unsafe_code": false,
|
|
"doc_to_choice": "def doc_to_choice(doc):\n idx = doc[\"sentence\"].index(\"_\")\n options = [doc[\"option1\"], doc[\"option2\"]]\n return [doc[\"sentence\"][:idx] + opt for opt in options]\n",
|
|
"description": "",
|
|
"target_delimiter": " ",
|
|
"fewshot_delimiter": "\n\n",
|
|
"fewshot_config": {
|
|
"sampler": "default",
|
|
"split": null,
|
|
"process_docs": null,
|
|
"fewshot_indices": null,
|
|
"samples": null,
|
|
"doc_to_text": "<function doc_to_text at 0x78a59400a340>",
|
|
"doc_to_choice": "<function doc_to_choice at 0x78a59400a480>",
|
|
"doc_to_target": "<function doc_to_target at 0x78a59400a3e0>",
|
|
"gen_prefix": null,
|
|
"fewshot_delimiter": "\n\n",
|
|
"target_delimiter": " "
|
|
},
|
|
"num_fewshot": 0,
|
|
"metric_list": [
|
|
{
|
|
"metric": "acc",
|
|
"aggregation": "mean",
|
|
"higher_is_better": true
|
|
}
|
|
],
|
|
"output_type": "multiple_choice",
|
|
"repeats": 1,
|
|
"should_decontaminate": true,
|
|
"doc_to_decontamination_query": "sentence",
|
|
"metadata": {
|
|
"version": 1.0,
|
|
"config_source": "/usr/local/lib/python3.12/dist-packages/lm_eval/tasks/winogrande/default.yaml"
|
|
}
|
|
}
|
|
},
|
|
"versions": {
|
|
"arc_easy": 1.0,
|
|
"boolq": 2.0,
|
|
"hellaswag": 1.0,
|
|
"lambada_openai": 1.0,
|
|
"piqa": 1.0,
|
|
"winogrande": 1.0
|
|
},
|
|
"n-shot": {
|
|
"arc_easy": 0,
|
|
"boolq": 0,
|
|
"hellaswag": 0,
|
|
"lambada_openai": 0,
|
|
"piqa": 0,
|
|
"winogrande": 0
|
|
},
|
|
"higher_is_better": {
|
|
"arc_easy": {
|
|
"acc": true,
|
|
"acc_norm": true
|
|
},
|
|
"boolq": {
|
|
"acc": true
|
|
},
|
|
"hellaswag": {
|
|
"acc": true,
|
|
"acc_norm": true
|
|
},
|
|
"lambada_openai": {
|
|
"perplexity": false,
|
|
"acc": true
|
|
},
|
|
"piqa": {
|
|
"acc": true,
|
|
"acc_norm": true
|
|
},
|
|
"winogrande": {
|
|
"acc": true
|
|
}
|
|
},
|
|
"n-samples": {
|
|
"arc_easy": {
|
|
"original": 2376,
|
|
"effective": 2376
|
|
},
|
|
"hellaswag": {
|
|
"original": 10042,
|
|
"effective": 10042
|
|
},
|
|
"piqa": {
|
|
"original": 1838,
|
|
"effective": 1838
|
|
},
|
|
"winogrande": {
|
|
"original": 1267,
|
|
"effective": 1267
|
|
},
|
|
"lambada_openai": {
|
|
"original": 5153,
|
|
"effective": 5153
|
|
},
|
|
"boolq": {
|
|
"original": 3270,
|
|
"effective": 3270
|
|
}
|
|
},
|
|
"config": {
|
|
"model": "./eval_ckpt",
|
|
"model_args": null,
|
|
"model_num_parameters": 31171072,
|
|
"model_dtype": "torch.float16",
|
|
"model_revision": "main",
|
|
"model_sha": "",
|
|
"batch_size": null,
|
|
"batch_sizes": [
|
|
64
|
|
],
|
|
"device": null,
|
|
"use_cache": null,
|
|
"limit": null,
|
|
"bootstrap_iters": 100000,
|
|
"gen_kwargs": null,
|
|
"random_seed": 0,
|
|
"numpy_seed": 1234,
|
|
"torch_seed": 1234,
|
|
"fewshot_seed": 1234
|
|
},
|
|
"git_hash": null,
|
|
"date": 1785579361.9488611,
|
|
"pretty_env_info": "PyTorch version: 2.11.0+cu128\nIs debug build: False\nCUDA used to build PyTorch: 12.8\nROCM used to build PyTorch: N/A\n\nOS: Ubuntu 22.04.5 LTS (x86_64)\nGCC version: (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0\nClang version: Could not collect\nCMake version: version 3.31.10\nLibc version: glibc-2.35\n\nPython version: 3.12.13 (main, Mar 4 2026, 09:23:07) [GCC 11.4.0] (64-bit runtime)\nPython platform: Linux-6.6.122+-x86_64-with-glibc2.35\nIs CUDA available: True\nCUDA runtime version: 12.8.93\nCUDA_MODULE_LOADING set to: \nGPU models and configuration: GPU 0: Tesla T4\nNvidia driver version: 580.82.07\ncuDNN version: Probably one of the following:\n/usr/lib/x86_64-linux-gnu/libcudnn.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_adv.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_cnn.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_precompiled.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_engines_runtime_compiled.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_graph.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_heuristic.so.9.8.0\n/usr/lib/x86_64-linux-gnu/libcudnn_ops.so.9.8.0\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nArchitecture: x86_64\nCPU op-mode(s): 32-bit, 64-bit\nAddress sizes: 46 bits physical, 48 bits virtual\nByte Order: Little Endian\nCPU(s): 2\nOn-line CPU(s) list: 0,1\nVendor ID: GenuineIntel\nModel name: Intel(R) Xeon(R) CPU @ 2.00GHz\nCPU family: 6\nModel: 85\nThread(s) per core: 2\nCore(s) per socket: 1\nSocket(s): 1\nStepping: 3\nBogoMIPS: 4000.30\nFlags: fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc rep_good nopl xtopology nonstop_tsc cpuid tsc_known_freq pni pclmulqdq ssse3 fma cx16 pcid sse4_1 sse4_2 x2apic movbe popcnt aes xsave avx f16c rdrand hypervisor lahf_lm abm 3dnowprefetch ssbd ibrs ibpb stibp fsgsbase tsc_adjust bmi1 hle avx2 smep bmi2 erms invpcid rtm mpx avx512f avx512dq rdseed adx smap clflushopt clwb avx512cd avx512bw avx512vl xsaveopt xsavec xgetbv1 xsaves arat md_clear arch_capabilities\nHypervisor vendor: KVM\nVirtualization type: full\nL1d cache: 32 KiB (1 instance)\nL1i cache: 32 KiB (1 instance)\nL2 cache: 1 MiB (1 instance)\nL3 cache: 38.5 MiB (1 instance)\nNUMA node(s): 1\nNUMA node0 CPU(s): 0,1\nVulnerability Gather data sampling: Not affected\nVulnerability Indirect target selection: Vulnerable\nVulnerability Itlb multihit: Not affected\nVulnerability L1tf: Mitigation; PTE Inversion\nVulnerability Mds: Vulnerable; SMT Host state unknown\nVulnerability Meltdown: Vulnerable\nVulnerability Mmio stale data: Vulnerable\nVulnerability Reg file data sampling: Not affected\nVulnerability Retbleed: Vulnerable\nVulnerability Spec rstack overflow: Not affected\nVulnerability Spec store bypass: Vulnerable\nVulnerability Spectre v1: Vulnerable: __user pointer sanitization and usercopy barriers only; no swapgs barriers\nVulnerability Spectre v2: Vulnerable; IBPB: disabled; STIBP: disabled; PBRSB-eIBRS: Not affected; BHI: Vulnerable\nVulnerability Srbds: Not affected\nVulnerability Tsa: Not affected\nVulnerability Tsx async abort: Vulnerable\nVulnerability Vmscape: Not affected\n\nVersions of relevant libraries:\n[pip3] intel-cmplr-lib-ur==2025.3.3\n[pip3] intel-openmp==2025.3.3\n[pip3] mkl==2025.3.1\n[pip3] numpy==2.0.2\n[pip3] nvidia-cublas-cu12==12.8.4.1\n[pip3] nvidia-cuda-cupti-cu12==12.8.90\n[pip3] nvidia-cuda-nvrtc-cu12==12.8.93\n[pip3] nvidia-cuda-runtime-cu12==12.8.90\n[pip3] nvidia-cudnn-cu12==9.19.0.56\n[pip3] nvidia-cufft-cu12==11.3.3.83\n[pip3] nvidia-curand-cu12==10.3.9.90\n[pip3] nvidia-cusolver-cu12==11.7.3.90\n[pip3] nvidia-cusparse-cu12==12.5.8.93\n[pip3] nvidia-cusparselt-cu12==0.7.1\n[pip3] nvidia-nccl-cu12==2.28.9\n[pip3] nvidia-nvjitlink-cu12==12.8.93\n[pip3] nvidia-nvtx-cu12==12.8.90\n[pip3] nvtx==0.2.15\n[pip3] onemkl-license==2025.3.1\n[pip3] optree==0.19.1\n[pip3] tbb==2022.3.1\n[pip3] tcmlib==1.5.0\n[pip3] torch==2.11.0+cu128\n[pip3] torchao==0.10.0\n[pip3] torchaudio==2.11.0+cu128\n[pip3] torchcodec==0.11.0+cu128\n[pip3] torchdata==0.11.0\n[pip3] torchsummary==1.5.1\n[pip3] torchtune==0.6.1\n[pip3] torchvision==0.26.0+cu128\n[pip3] triton==3.6.0\n[pip3] umf==1.0.3\n[conda] Could not collect",
|
|
"transformers_version": "5.13.1",
|
|
"lm_eval_version": "0.4.12",
|
|
"upper_git_hash": null,
|
|
"tokenizer_pad_token": [
|
|
"<pad>",
|
|
"0"
|
|
],
|
|
"tokenizer_eos_token": [
|
|
"</s>",
|
|
"3"
|
|
],
|
|
"tokenizer_bos_token": [
|
|
"<s>",
|
|
"2"
|
|
],
|
|
"eot_token_id": 3,
|
|
"max_length": 1024
|
|
} |