初始化项目,由ModelHub XC社区提供模型
Model: Yaseal/llama3_3b_instruct_vallina_full_sft_30k Source: Original Platform
This commit is contained in:
57
.gitattributes
vendored
Normal file
57
.gitattributes
vendored
Normal file
@@ -0,0 +1,57 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052/predictions/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052/reviews/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134/predictions/llama3_3b_instruct_vallina_full_sft_30k/agieval_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511/predictions/llama3_3b_instruct_vallina_full_sft_30k/gsm8k_main.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/AIME24/20260313_124640/predictions/llama3_3b_instruct_vallina_full_sft_30k/aime24_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/GPQA/20260314_110837/predictions/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_1/GPQA/20260314_110837/reviews/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/MMLUProNoMath/20260315_123641/predictions/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/MMLUProNoMath/20260315_123641/reviews/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/AGIEvalMath/20260315_081307/predictions/llama3_3b_instruct_vallina_full_sft_30k/agieval_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/GSM8K/20260315_100259/predictions/llama3_3b_instruct_vallina_full_sft_30k/gsm8k_main.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/AIME24/20260315_061730/predictions/llama3_3b_instruct_vallina_full_sft_30k/aime24_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/GPQA/20260315_113454/predictions/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_3/GPQA/20260315_113454/reviews/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/MMLUProNoMath/20260315_024629/predictions/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/MMLUProNoMath/20260315_024629/reviews/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/AGIEvalMath/20260314_222836/predictions/llama3_3b_instruct_vallina_full_sft_30k/agieval_math_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/GSM8K/20260315_001859/predictions/llama3_3b_instruct_vallina_full_sft_30k/gsm8k_main.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/AIME24/20260314_203234/predictions/llama3_3b_instruct_vallina_full_sft_30k/aime24_default.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/GPQA/20260315_014354/predictions/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
evalscope/version_20260313_110435/run_2/GPQA/20260315_014354/reviews/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend_gpqa.jsonl filter=lfs diff=lfs merge=lfs -text
|
||||
66
README.md
Normal file
66
README.md
Normal file
@@ -0,0 +1,66 @@
|
||||
---
|
||||
library_name: transformers
|
||||
license: other
|
||||
base_model: LLM-Research/Llama-3.2-3B-Instruct
|
||||
tags:
|
||||
- llama-factory
|
||||
- full
|
||||
- generated_from_trainer
|
||||
model-index:
|
||||
- name: llama3_3b_instruct_vallina_full_sft_30k
|
||||
results: []
|
||||
---
|
||||
|
||||
<!-- This model card has been generated automatically according to the information the Trainer had access to. You
|
||||
should probably proofread and complete it, then remove this comment. -->
|
||||
|
||||
# llama3_3b_instruct_vallina_full_sft_30k
|
||||
|
||||
This model is a fine-tuned version of [LLM-Research/Llama-3.2-3B-Instruct](https://huggingface.co/LLM-Research/Llama-3.2-3B-Instruct) on the deepmath_plain_30k_train dataset.
|
||||
It achieves the following results on the evaluation set:
|
||||
- Loss: 0.4737
|
||||
|
||||
## Model description
|
||||
|
||||
More information needed
|
||||
|
||||
## Intended uses & limitations
|
||||
|
||||
More information needed
|
||||
|
||||
## Training and evaluation data
|
||||
|
||||
More information needed
|
||||
|
||||
## Training procedure
|
||||
|
||||
### Training hyperparameters
|
||||
|
||||
The following hyperparameters were used during training:
|
||||
- learning_rate: 2e-05
|
||||
- train_batch_size: 1
|
||||
- eval_batch_size: 1
|
||||
- seed: 42
|
||||
- distributed_type: multi-GPU
|
||||
- num_devices: 2
|
||||
- gradient_accumulation_steps: 8
|
||||
- total_train_batch_size: 16
|
||||
- total_eval_batch_size: 2
|
||||
- optimizer: Use OptimizerNames.ADAMW_TORCH with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
|
||||
- lr_scheduler_type: cosine
|
||||
- lr_scheduler_warmup_ratio: 0.1
|
||||
- num_epochs: 2.0
|
||||
|
||||
### Training results
|
||||
|
||||
| Training Loss | Epoch | Step | Validation Loss |
|
||||
|:-------------:|:------:|:----:|:---------------:|
|
||||
| 0.3879 | 1.7182 | 1000 | 0.4751 |
|
||||
|
||||
|
||||
### Framework versions
|
||||
|
||||
- Transformers 4.52.4
|
||||
- Pytorch 2.6.0+cu124
|
||||
- Datasets 3.6.0
|
||||
- Tokenizers 0.21.1
|
||||
12
all_results.json
Normal file
12
all_results.json
Normal file
@@ -0,0 +1,12 @@
|
||||
{
|
||||
"epoch": 2.0,
|
||||
"eval_loss": 0.47373512387275696,
|
||||
"eval_runtime": 21.6176,
|
||||
"eval_samples_per_second": 4.996,
|
||||
"eval_steps_per_second": 2.498,
|
||||
"total_flos": 2.8348028555949507e+18,
|
||||
"train_loss": 0.47038160638301235,
|
||||
"train_runtime": 14726.6059,
|
||||
"train_samples_per_second": 1.265,
|
||||
"train_steps_per_second": 0.079
|
||||
}
|
||||
93
chat_template.jinja
Normal file
93
chat_template.jinja
Normal file
@@ -0,0 +1,93 @@
|
||||
{{- bos_token }}
|
||||
{%- if custom_tools is defined %}
|
||||
{%- set tools = custom_tools %}
|
||||
{%- endif %}
|
||||
{%- if not tools_in_user_message is defined %}
|
||||
{%- set tools_in_user_message = true %}
|
||||
{%- endif %}
|
||||
{%- if not date_string is defined %}
|
||||
{%- if strftime_now is defined %}
|
||||
{%- set date_string = strftime_now("%d %b %Y") %}
|
||||
{%- else %}
|
||||
{%- set date_string = "26 Jul 2024" %}
|
||||
{%- endif %}
|
||||
{%- endif %}
|
||||
{%- if not tools is defined %}
|
||||
{%- set tools = none %}
|
||||
{%- endif %}
|
||||
|
||||
{#- This block extracts the system message, so we can slot it into the right place. #}
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{%- set system_message = messages[0]['content']|trim %}
|
||||
{%- set messages = messages[1:] %}
|
||||
{%- else %}
|
||||
{%- set system_message = "" %}
|
||||
{%- endif %}
|
||||
|
||||
{#- System message #}
|
||||
{{- "<|start_header_id|>system<|end_header_id|>\n\n" }}
|
||||
{%- if tools is not none %}
|
||||
{{- "Environment: ipython\n" }}
|
||||
{%- endif %}
|
||||
{{- "Cutting Knowledge Date: December 2023\n" }}
|
||||
{{- "Today Date: " + date_string + "\n\n" }}
|
||||
{%- if tools is not none and not tools_in_user_message %}
|
||||
{{- "You have access to the following functions. To call a function, please respond with JSON for a function call." }}
|
||||
{{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }}
|
||||
{{- "Do not use variables.\n\n" }}
|
||||
{%- for t in tools %}
|
||||
{{- t | tojson(indent=4) }}
|
||||
{{- "\n\n" }}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{{- system_message }}
|
||||
{{- "<|eot_id|>" }}
|
||||
|
||||
{#- Custom tools are passed in a user message with some extra guidance #}
|
||||
{%- if tools_in_user_message and not tools is none %}
|
||||
{#- Extract the first user message so we can plug it in here #}
|
||||
{%- if messages | length != 0 %}
|
||||
{%- set first_user_message = messages[0]['content']|trim %}
|
||||
{%- set messages = messages[1:] %}
|
||||
{%- else %}
|
||||
{{- raise_exception("Cannot put tools in the first user message when there's no first user message!") }}
|
||||
{%- endif %}
|
||||
{{- '<|start_header_id|>user<|end_header_id|>\n\n' -}}
|
||||
{{- "Given the following functions, please respond with a JSON for a function call " }}
|
||||
{{- "with its proper arguments that best answers the given prompt.\n\n" }}
|
||||
{{- 'Respond in the format {"name": function name, "parameters": dictionary of argument name and its value}.' }}
|
||||
{{- "Do not use variables.\n\n" }}
|
||||
{%- for t in tools %}
|
||||
{{- t | tojson(indent=4) }}
|
||||
{{- "\n\n" }}
|
||||
{%- endfor %}
|
||||
{{- first_user_message + "<|eot_id|>"}}
|
||||
{%- endif %}
|
||||
|
||||
{%- for message in messages %}
|
||||
{%- if not (message.role == 'ipython' or message.role == 'tool' or 'tool_calls' in message) %}
|
||||
{{- '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' }}
|
||||
{%- elif 'tool_calls' in message %}
|
||||
{%- if not message.tool_calls|length == 1 %}
|
||||
{{- raise_exception("This model only supports single tool-calls at once!") }}
|
||||
{%- endif %}
|
||||
{%- set tool_call = message.tool_calls[0].function %}
|
||||
{{- '<|start_header_id|>assistant<|end_header_id|>\n\n' -}}
|
||||
{{- '{"name": "' + tool_call.name + '", ' }}
|
||||
{{- '"parameters": ' }}
|
||||
{{- tool_call.arguments | tojson }}
|
||||
{{- "}" }}
|
||||
{{- "<|eot_id|>" }}
|
||||
{%- elif message.role == "tool" or message.role == "ipython" %}
|
||||
{{- "<|start_header_id|>ipython<|end_header_id|>\n\n" }}
|
||||
{%- if message.content is mapping or message.content is iterable %}
|
||||
{{- message.content | tojson }}
|
||||
{%- else %}
|
||||
{{- message.content }}
|
||||
{%- endif %}
|
||||
{{- "<|eot_id|>" }}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- if add_generation_prompt %}
|
||||
{{- '<|start_header_id|>assistant<|end_header_id|>\n\n' }}
|
||||
{%- endif %}
|
||||
39
config.json
Normal file
39
config.json
Normal file
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"architectures": [
|
||||
"LlamaForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 128000,
|
||||
"eos_token_id": [
|
||||
128001,
|
||||
128008,
|
||||
128009
|
||||
],
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 3072,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 8192,
|
||||
"max_position_embeddings": 131072,
|
||||
"mlp_bias": false,
|
||||
"model_type": "llama",
|
||||
"num_attention_heads": 24,
|
||||
"num_hidden_layers": 28,
|
||||
"num_key_value_heads": 8,
|
||||
"pretraining_tp": 1,
|
||||
"rms_norm_eps": 1e-05,
|
||||
"rope_scaling": {
|
||||
"factor": 32.0,
|
||||
"high_freq_factor": 4.0,
|
||||
"low_freq_factor": 1.0,
|
||||
"original_max_position_embeddings": 8192,
|
||||
"rope_type": "llama3"
|
||||
},
|
||||
"rope_theta": 500000.0,
|
||||
"tie_word_embeddings": true,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.52.4",
|
||||
"use_cache": false,
|
||||
"vocab_size": 128256
|
||||
}
|
||||
7
eval_results.json
Normal file
7
eval_results.json
Normal file
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"epoch": 2.0,
|
||||
"eval_loss": 0.47373512387275696,
|
||||
"eval_runtime": 21.6176,
|
||||
"eval_samples_per_second": 4.996,
|
||||
"eval_steps_per_second": 2.498
|
||||
}
|
||||
371
evalscope/version_20260313_110435/multi_run_summary.json
Normal file
371
evalscope/version_20260313_110435/multi_run_summary.json
Normal file
@@ -0,0 +1,371 @@
|
||||
{
|
||||
"num_runs": 3,
|
||||
"version": "version_20260313_110435",
|
||||
"config_file": "eval_configs/Qwen2.5-7B-Instruct.yaml",
|
||||
"results": [
|
||||
{
|
||||
"benchmark": "AIME25",
|
||||
"dataset": "aime25/AIME2025-I",
|
||||
"run_count": 3,
|
||||
"mean": 0.0167,
|
||||
"std": 0.0084,
|
||||
"std_percent": 50.1,
|
||||
"min": 0.0083,
|
||||
"max": 0.025,
|
||||
"range": 0.0167,
|
||||
"run_1": 0.0167,
|
||||
"run_2": 0.025,
|
||||
"run_3": 0.0083
|
||||
},
|
||||
{
|
||||
"benchmark": "AIME25",
|
||||
"dataset": "aime25/AIME2025-II",
|
||||
"run_count": 3,
|
||||
"mean": 0.0056,
|
||||
"std": 0.0096,
|
||||
"std_percent": 173.21,
|
||||
"min": 0.0,
|
||||
"max": 0.0167,
|
||||
"range": 0.0167,
|
||||
"run_1": 0.0167,
|
||||
"run_2": 0.0,
|
||||
"run_3": 0.0
|
||||
},
|
||||
{
|
||||
"benchmark": "AIME25",
|
||||
"dataset": "aime25/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.0111,
|
||||
"std": 0.0064,
|
||||
"std_percent": 57.14,
|
||||
"min": 0.0042,
|
||||
"max": 0.0167,
|
||||
"range": 0.0125,
|
||||
"run_1": 0.0167,
|
||||
"run_2": 0.0125,
|
||||
"run_3": 0.0042
|
||||
},
|
||||
{
|
||||
"benchmark": "GSM8K",
|
||||
"dataset": "gsm8k/main",
|
||||
"run_count": 3,
|
||||
"mean": 0.7526,
|
||||
"std": 0.0137,
|
||||
"std_percent": 1.81,
|
||||
"min": 0.7392,
|
||||
"max": 0.7665,
|
||||
"range": 0.0273,
|
||||
"run_1": 0.7521,
|
||||
"run_2": 0.7665,
|
||||
"run_3": 0.7392
|
||||
},
|
||||
{
|
||||
"benchmark": "GSM8K",
|
||||
"dataset": "gsm8k/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.7526,
|
||||
"std": 0.0137,
|
||||
"std_percent": 1.81,
|
||||
"min": 0.7392,
|
||||
"max": 0.7665,
|
||||
"range": 0.0273,
|
||||
"run_1": 0.7521,
|
||||
"run_2": 0.7665,
|
||||
"run_3": 0.7392
|
||||
},
|
||||
{
|
||||
"benchmark": "AGIEvalMath",
|
||||
"dataset": "agieval_math/default",
|
||||
"run_count": 3,
|
||||
"mean": 0.4587,
|
||||
"std": 0.0115,
|
||||
"std_percent": 2.52,
|
||||
"min": 0.452,
|
||||
"max": 0.472,
|
||||
"range": 0.02,
|
||||
"run_1": 0.472,
|
||||
"run_2": 0.452,
|
||||
"run_3": 0.452
|
||||
},
|
||||
{
|
||||
"benchmark": "AGIEvalMath",
|
||||
"dataset": "agieval_math/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.4587,
|
||||
"std": 0.0115,
|
||||
"std_percent": 2.52,
|
||||
"min": 0.452,
|
||||
"max": 0.472,
|
||||
"range": 0.02,
|
||||
"run_1": 0.472,
|
||||
"run_2": 0.452,
|
||||
"run_3": 0.452
|
||||
},
|
||||
{
|
||||
"benchmark": "AMC",
|
||||
"dataset": "amc/amc22",
|
||||
"run_count": 3,
|
||||
"mean": 0.1318,
|
||||
"std": 0.0268,
|
||||
"std_percent": 20.37,
|
||||
"min": 0.1163,
|
||||
"max": 0.1628,
|
||||
"range": 0.0465,
|
||||
"run_1": 0.1163,
|
||||
"run_2": 0.1628,
|
||||
"run_3": 0.1163
|
||||
},
|
||||
{
|
||||
"benchmark": "AMC",
|
||||
"dataset": "amc/amc23",
|
||||
"run_count": 3,
|
||||
"mean": 0.1739,
|
||||
"std": 0.0784,
|
||||
"std_percent": 45.05,
|
||||
"min": 0.087,
|
||||
"max": 0.2391,
|
||||
"range": 0.1521,
|
||||
"run_1": 0.2391,
|
||||
"run_2": 0.087,
|
||||
"run_3": 0.1957
|
||||
},
|
||||
{
|
||||
"benchmark": "AMC",
|
||||
"dataset": "amc/amc24",
|
||||
"run_count": 3,
|
||||
"mean": 0.1333,
|
||||
"std": 0.0445,
|
||||
"std_percent": 33.34,
|
||||
"min": 0.0889,
|
||||
"max": 0.1778,
|
||||
"range": 0.0889,
|
||||
"run_1": 0.1333,
|
||||
"run_2": 0.1778,
|
||||
"run_3": 0.0889
|
||||
},
|
||||
{
|
||||
"benchmark": "AMC",
|
||||
"dataset": "amc/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.1468,
|
||||
"std": 0.0155,
|
||||
"std_percent": 10.57,
|
||||
"min": 0.1344,
|
||||
"max": 0.1642,
|
||||
"range": 0.0298,
|
||||
"run_1": 0.1642,
|
||||
"run_2": 0.1418,
|
||||
"run_3": 0.1344
|
||||
},
|
||||
{
|
||||
"benchmark": "MMLUProNoMath",
|
||||
"dataset": "mmlu_pro_no_math/default",
|
||||
"run_count": 3,
|
||||
"mean": 0.3513,
|
||||
"std": 0.0017,
|
||||
"std_percent": 0.48,
|
||||
"min": 0.3496,
|
||||
"max": 0.353,
|
||||
"range": 0.0034,
|
||||
"run_1": 0.353,
|
||||
"run_2": 0.3496,
|
||||
"run_3": 0.3513
|
||||
},
|
||||
{
|
||||
"benchmark": "MMLUProNoMath",
|
||||
"dataset": "mmlu_pro_no_math/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.3513,
|
||||
"std": 0.0017,
|
||||
"std_percent": 0.48,
|
||||
"min": 0.3496,
|
||||
"max": 0.353,
|
||||
"range": 0.0034,
|
||||
"run_1": 0.353,
|
||||
"run_2": 0.3496,
|
||||
"run_3": 0.3513
|
||||
},
|
||||
{
|
||||
"benchmark": "AIME24",
|
||||
"dataset": "aime24/default",
|
||||
"run_count": 3,
|
||||
"mean": 0.0181,
|
||||
"std": 0.0024,
|
||||
"std_percent": 13.1,
|
||||
"min": 0.0167,
|
||||
"max": 0.0208,
|
||||
"range": 0.0041,
|
||||
"run_1": 0.0167,
|
||||
"run_2": 0.0167,
|
||||
"run_3": 0.0208
|
||||
},
|
||||
{
|
||||
"benchmark": "AIME24",
|
||||
"dataset": "aime24/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.0181,
|
||||
"std": 0.0024,
|
||||
"std_percent": 13.1,
|
||||
"min": 0.0167,
|
||||
"max": 0.0208,
|
||||
"range": 0.0041,
|
||||
"run_1": 0.0167,
|
||||
"run_2": 0.0167,
|
||||
"run_3": 0.0208
|
||||
},
|
||||
{
|
||||
"benchmark": "GPQA",
|
||||
"dataset": "gpqa_extend/gpqa",
|
||||
"run_count": 3,
|
||||
"mean": 0.2375,
|
||||
"std": 0.0204,
|
||||
"std_percent": 8.59,
|
||||
"min": 0.2143,
|
||||
"max": 0.2527,
|
||||
"range": 0.0384,
|
||||
"run_1": 0.2454,
|
||||
"run_2": 0.2527,
|
||||
"run_3": 0.2143
|
||||
},
|
||||
{
|
||||
"benchmark": "GPQA",
|
||||
"dataset": "gpqa_extend/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.2375,
|
||||
"std": 0.0204,
|
||||
"std_percent": 8.59,
|
||||
"min": 0.2143,
|
||||
"max": 0.2527,
|
||||
"range": 0.0384,
|
||||
"run_1": 0.2454,
|
||||
"run_2": 0.2527,
|
||||
"run_3": 0.2143
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/Level 1",
|
||||
"run_count": 3,
|
||||
"mean": 0.7442,
|
||||
"std": 0.0233,
|
||||
"std_percent": 3.12,
|
||||
"min": 0.7209,
|
||||
"max": 0.7674,
|
||||
"range": 0.0465,
|
||||
"run_1": 0.7442,
|
||||
"run_2": 0.7209,
|
||||
"run_3": 0.7674
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/Level 2",
|
||||
"run_count": 3,
|
||||
"mean": 0.7111,
|
||||
"std": 0.0294,
|
||||
"std_percent": 4.13,
|
||||
"min": 0.6778,
|
||||
"max": 0.7333,
|
||||
"range": 0.0555,
|
||||
"run_1": 0.7222,
|
||||
"run_2": 0.7333,
|
||||
"run_3": 0.6778
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/Level 3",
|
||||
"run_count": 3,
|
||||
"mean": 0.5587,
|
||||
"std": 0.0145,
|
||||
"std_percent": 2.6,
|
||||
"min": 0.5429,
|
||||
"max": 0.5714,
|
||||
"range": 0.0285,
|
||||
"run_1": 0.5714,
|
||||
"run_2": 0.5429,
|
||||
"run_3": 0.5619
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/Level 4",
|
||||
"run_count": 3,
|
||||
"mean": 0.375,
|
||||
"std": 0.0234,
|
||||
"std_percent": 6.24,
|
||||
"min": 0.3516,
|
||||
"max": 0.3984,
|
||||
"range": 0.0468,
|
||||
"run_1": 0.375,
|
||||
"run_2": 0.3984,
|
||||
"run_3": 0.3516
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/Level 5",
|
||||
"run_count": 3,
|
||||
"mean": 0.1891,
|
||||
"std": 0.0215,
|
||||
"std_percent": 11.39,
|
||||
"min": 0.1642,
|
||||
"max": 0.2015,
|
||||
"range": 0.0373,
|
||||
"run_1": 0.2015,
|
||||
"run_2": 0.1642,
|
||||
"run_3": 0.2015
|
||||
},
|
||||
{
|
||||
"benchmark": "MATH500",
|
||||
"dataset": "math_500/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.456,
|
||||
"std": 0.0072,
|
||||
"std_percent": 1.58,
|
||||
"min": 0.45,
|
||||
"max": 0.464,
|
||||
"range": 0.014,
|
||||
"run_1": 0.464,
|
||||
"run_2": 0.454,
|
||||
"run_3": 0.45
|
||||
},
|
||||
{
|
||||
"benchmark": "ARC",
|
||||
"dataset": "arc/ARC-Easy",
|
||||
"run_count": 3,
|
||||
"mean": 0.8173,
|
||||
"std": 0.009,
|
||||
"std_percent": 1.1,
|
||||
"min": 0.8102,
|
||||
"max": 0.8274,
|
||||
"range": 0.0172,
|
||||
"run_1": 0.8274,
|
||||
"run_2": 0.8144,
|
||||
"run_3": 0.8102
|
||||
},
|
||||
{
|
||||
"benchmark": "ARC",
|
||||
"dataset": "arc/ARC-Challenge",
|
||||
"run_count": 3,
|
||||
"mean": 0.723,
|
||||
"std": 0.0071,
|
||||
"std_percent": 0.99,
|
||||
"min": 0.715,
|
||||
"max": 0.7287,
|
||||
"range": 0.0137,
|
||||
"run_1": 0.715,
|
||||
"run_2": 0.7287,
|
||||
"run_3": 0.7253
|
||||
},
|
||||
{
|
||||
"benchmark": "ARC",
|
||||
"dataset": "arc/all",
|
||||
"run_count": 3,
|
||||
"mean": 0.7862,
|
||||
"std": 0.0041,
|
||||
"std_percent": 0.52,
|
||||
"min": 0.7822,
|
||||
"max": 0.7903,
|
||||
"range": 0.0081,
|
||||
"run_1": 0.7903,
|
||||
"run_2": 0.7861,
|
||||
"run_3": 0.7822
|
||||
}
|
||||
]
|
||||
}
|
||||
BIN
evalscope/version_20260313_110435/multi_run_summary.xlsx
Normal file
BIN
evalscope/version_20260313_110435/multi_run_summary.xlsx
Normal file
Binary file not shown.
6
evalscope/version_20260313_110435/progress.json
Normal file
6
evalscope/version_20260313_110435/progress.json
Normal file
@@ -0,0 +1,6 @@
|
||||
{
|
||||
"version": "version_20260313_110435",
|
||||
"num_runs": 3,
|
||||
"config_file": "eval_configs/Qwen2.5-7B-Instruct.yaml",
|
||||
"created_at": "2026-03-16T10:25:51.462175"
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
agieval_math:
|
||||
aggregation: mean
|
||||
dataset_id: hails/agieval-math
|
||||
default_subset: default
|
||||
description: AGIEval-Math is a subset of AGIEval containing 1000 competition-level
|
||||
math problems drawn from the MATH dataset, covering algebra, geometry, number
|
||||
theory and more. Answers are clean numerical or symbolic expressions.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: agieval_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AGIEval-Math
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: Please reason step by step to solve the problem. Put your final
|
||||
answer in \boxed{}.
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- agieval_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
@@ -0,0 +1,78 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
agieval_math:
|
||||
aggregation: mean
|
||||
dataset_id: hails/agieval-math
|
||||
default_subset: default
|
||||
description: AGIEval-Math is a subset of AGIEval containing 1000 competition-level
|
||||
math problems drawn from the MATH dataset, covering algebra, geometry, number
|
||||
theory and more. Answers are clean numerical or symbolic expressions.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: agieval_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AGIEval-Math
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: Please reason step by step to solve the problem. Put your final
|
||||
answer in \boxed{}.
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- agieval_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
@@ -0,0 +1,78 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
agieval_math:
|
||||
aggregation: mean
|
||||
dataset_id: hails/agieval-math
|
||||
default_subset: default
|
||||
description: AGIEval-Math is a subset of AGIEval containing 1000 competition-level
|
||||
math problems drawn from the MATH dataset, covering algebra, geometry, number
|
||||
theory and more. Answers are clean numerical or symbolic expressions.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: agieval_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AGIEval-Math
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: Please reason step by step to solve the problem. Put your final
|
||||
answer in \boxed{}.
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- agieval_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
@@ -0,0 +1,180 @@
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134/configs/task_config_b7926b.yaml
|
||||
2026-03-13 22:35:35 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"agieval_math"
|
||||
],
|
||||
"dataset_args": {
|
||||
"agieval_math": {
|
||||
"name": "agieval_math",
|
||||
"dataset_id": "hails/agieval-math",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "test",
|
||||
"prompt_template": "{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": "Please reason step by step to solve the problem. Put your final answer in \\boxed{}.",
|
||||
"query_template": null,
|
||||
"pretty_name": "AGIEval-Math",
|
||||
"description": "AGIEval-Math is a subset of AGIEval containing 1000 competition-level math problems drawn from the MATH dataset, covering algebra, geometry, number theory and more. Answers are clean numerical or symbolic expressions.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134",
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Start loading benchmark dataset: agieval_math
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Start evaluating 1 subsets of the agieval_math: ['default']
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Evaluating subset: default
|
||||
2026-03-13 22:35:35 - evalscope - INFO: Getting predictions for subset: default
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Reusing predictions from /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134/predictions/llama3_3b_instruct_vallina_full_sft_30k/agieval_math_default.jsonl, got 839 predictions, remaining 161 samples
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Processing 161 samples, if data is large, it may take a while.
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=3, inflight=1)
|
||||
2026-03-13 22:35:36 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 22:36:36 - evalscope - INFO: Predicting[agieval_math@default]: 0%| 0/161 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 22:37:09 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=149, inflight=2)
|
||||
2026-03-13 22:37:18 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=141, inflight=2)
|
||||
2026-03-13 22:37:36 - evalscope - INFO: Predicting[agieval_math@default]: 2%| 4/161 [Elapsed: 02:00 < Remaining: 1:11:51, 27.46s/it]
|
||||
2026-03-13 22:38:36 - evalscope - INFO: Predicting[agieval_math@default]: 2%| 4/161 [Elapsed: 03:00 < Remaining: 1:11:51, 27.46s/it]
|
||||
2026-03-13 22:38:59 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=133, inflight=2)
|
||||
2026-03-13 22:39:00 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=125, inflight=2)
|
||||
2026-03-13 22:39:36 - evalscope - INFO: Predicting[agieval_math@default]: 12%| 20/161 [Elapsed: 04:00 < Remaining: 1:33:00, 39.58s/it]
|
||||
2026-03-13 22:40:36 - evalscope - INFO: Predicting[agieval_math@default]: 12%| 20/161 [Elapsed: 05:00 < Remaining: 1:33:00, 39.58s/it]
|
||||
2026-03-13 22:40:42 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=117, inflight=2)
|
||||
2026-03-13 22:40:52 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=109, inflight=2)
|
||||
2026-03-13 22:41:36 - evalscope - INFO: Predicting[agieval_math@default]: 22%| 36/161 [Elapsed: 06:00 < Remaining: 15:29, 7.43s/it]
|
||||
2026-03-13 22:42:35 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=101, inflight=2)
|
||||
2026-03-13 22:42:36 - evalscope - INFO: Predicting[agieval_math@default]: 27%| 44/161 [Elapsed: 07:00 < Remaining: 18:13, 9.35s/it]
|
||||
2026-03-13 22:42:38 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=93, inflight=2)
|
||||
2026-03-13 22:43:36 - evalscope - INFO: Predicting[agieval_math@default]: 32%| 52/161 [Elapsed: 08:00 < Remaining: 11:36, 6.39s/it]
|
||||
2026-03-13 22:44:26 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=85, inflight=2)
|
||||
2026-03-13 22:44:28 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=77, inflight=2)
|
||||
2026-03-13 22:44:36 - evalscope - INFO: Predicting[agieval_math@default]: 42%| 68/161 [Elapsed: 09:00 < Remaining: 09:16, 5.99s/it]
|
||||
2026-03-13 22:45:36 - evalscope - INFO: Predicting[agieval_math@default]: 42%| 68/161 [Elapsed: 10:00 < Remaining: 09:16, 5.99s/it]
|
||||
2026-03-13 22:46:02 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=69, inflight=2)
|
||||
2026-03-13 22:46:09 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=61, inflight=2)
|
||||
2026-03-13 22:46:36 - evalscope - INFO: Predicting[agieval_math@default]: 52%| 84/161 [Elapsed: 11:00 < Remaining: 07:18, 5.70s/it]
|
||||
2026-03-13 22:47:04 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=53, inflight=2)
|
||||
2026-03-13 22:47:36 - evalscope - INFO: Predicting[agieval_math@default]: 57%| 92/161 [Elapsed: 12:00 < Remaining: 06:55, 6.02s/it]
|
||||
2026-03-13 22:48:03 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=45, inflight=2)
|
||||
2026-03-13 22:48:36 - evalscope - INFO: Predicting[agieval_math@default]: 62%| 100/161 [Elapsed: 13:00 < Remaining: 06:32, 6.43s/it]
|
||||
2026-03-13 22:48:45 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=37, inflight=2)
|
||||
2026-03-13 22:49:36 - evalscope - INFO: Predicting[agieval_math@default]: 67%| 108/161 [Elapsed: 14:00 < Remaining: 05:21, 6.08s/it]
|
||||
2026-03-13 22:49:50 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=29, inflight=2)
|
||||
2026-03-13 22:50:37 - evalscope - INFO: Predicting[agieval_math@default]: 72%| 116/161 [Elapsed: 15:00 < Remaining: 05:01, 6.69s/it]
|
||||
2026-03-13 22:50:41 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=21, inflight=2)
|
||||
2026-03-13 22:51:35 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=13, inflight=2)
|
||||
2026-03-13 22:51:37 - evalscope - INFO: Predicting[agieval_math@default]: 82%| 132/161 [Elapsed: 16:00 < Remaining: 03:11, 6.61s/it]
|
||||
2026-03-13 22:52:37 - evalscope - INFO: Predicting[agieval_math@default]: 82%| 132/161 [Elapsed: 17:00 < Remaining: 03:11, 6.61s/it]
|
||||
2026-03-13 22:52:47 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=5, inflight=2)
|
||||
2026-03-13 22:53:17 - evalscope - INFO: Dispatcher: Worker-1 <- 5 prompts (pending=0, inflight=2)
|
||||
2026-03-13 22:53:37 - evalscope - INFO: Predicting[agieval_math@default]: 92%| 148/161 [Elapsed: 18:00 < Remaining: 01:21, 6.28s/it]
|
||||
2026-03-13 22:54:37 - evalscope - INFO: Predicting[agieval_math@default]: 97%| 156/161 [Elapsed: 19:00 < Remaining: 00:35, 7.17s/it]
|
||||
2026-03-13 22:54:59 - evalscope - INFO: Predicting[agieval_math@default]: 100%| 161/161 [Elapsed: 19:22 < Remaining: 00:00, 6.04s/it]
|
||||
2026-03-13 22:54:59 - evalscope - INFO: Finished getting predictions for subset: default.
|
||||
2026-03-13 22:54:59 - evalscope - INFO: Getting reviews for subset: default
|
||||
2026-03-13 22:54:59 - evalscope - INFO: Reviewing 1000 samples, if data is large, it may take a while.
|
||||
2026-03-13 22:55:10 - evalscope - INFO: Reviewing[agieval_math@default]: 100%| 1000/1000 [Elapsed: 00:11 < Remaining: 00:00, 39.72it/s]
|
||||
2026-03-13 22:55:10 - evalscope - INFO: Finished reviewing subset: default. Total reviewed: 1000
|
||||
2026-03-13 22:55:10 - evalscope - INFO: Aggregating scores for subset: default
|
||||
2026-03-13 22:55:10 - evalscope - INFO: Evaluating [agieval_math] 100%| 1/1 [Elapsed: 19:35 < Remaining: 00:00, 1175.27s/subset]
|
||||
2026-03-13 22:55:10 - evalscope - INFO: Generating report...
|
||||
2026-03-13 22:55:11 - evalscope - INFO:
|
||||
agieval_math report table:
|
||||
+-----------------------------------------+--------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+==============+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | agieval_math | mean_acc | default | 1000 | 0.472 | default |
|
||||
+-----------------------------------------+--------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134/reports/llama3_3b_instruct_vallina_full_sft_30k/agieval_math.json
|
||||
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Benchmark agieval_math evaluation finished.
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+--------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+==============+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | agieval_math | mean_acc | default | 1000 | 0.472 | default |
|
||||
+-----------------------------------------+--------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['agieval_math']
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath/20260313_173134
|
||||
2026-03-13 22:55:11 - evalscope - INFO: [进度条] AGIEvalMath 评测完成 ✓
|
||||
2026-03-13 22:55:11 - evalscope - INFO: [断点续传] AGIEvalMath 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AGIEvalMath_result.json
|
||||
2026-03-13 22:55:11 - evalscope - INFO: 完成评测 AGIEvalMath (5/8)
|
||||
2026-03-13 22:55:11 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-13 22:55:11 - evalscope - INFO: 正在评估 GSM8K (repeat: 1次) (剩余: 2个)
|
||||
2026-03-13 22:55:11 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-13 22:55:11 - evalscope - INFO: 开始创建 benchmark GSM8K 的 TaskConfig
|
||||
2026-03-13 22:55:11 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-13 22:55:11 - evalscope - INFO: [GSM8K] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['gsm8k'], eval_batch_size=2048
|
||||
2026-03-13 22:55:11 - evalscope - INFO: 开始评测 GSM8K...
|
||||
2026-03-13 22:55:11 - evalscope - INFO: [进度条] GSM8K 开始评测
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:170a13287f46ec31061957bd75cbe6469d954756881485926abc254904b0b318
|
||||
size 38051086
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@agieval_math",
|
||||
"dataset_name": "agieval_math",
|
||||
"dataset_pretty_name": "AGIEval-Math",
|
||||
"dataset_description": "AGIEval-Math is a subset of AGIEval containing 1000 competition-level math problems drawn from the MATH dataset, covering algebra, geometry, number theory and more. Answers are clean numerical or symbolic expressions.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.472,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.472,
|
||||
"num": 1000
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"agieval_math": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@agieval_math",
|
||||
"dataset_name": "agieval_math",
|
||||
"dataset_pretty_name": "AGIEval-Math",
|
||||
"dataset_description": "AGIEval-Math is a subset of AGIEval containing 1000 competition-level math problems drawn from the MATH dataset, covering algebra, geometry, number theory and more. Answers are clean numerical or symbolic expressions.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.472,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.472,
|
||||
"num": 1000
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
aime24:
|
||||
aggregation: mean
|
||||
dataset_id: HuggingFaceH4/aime_2024
|
||||
default_subset: default
|
||||
description: The AIME 2024 benchmark is based on problems from the American Invitational
|
||||
Mathematics Examination, a prestigious high school mathematics competition.
|
||||
This benchmark tests a model's ability to solve challenging mathematics problems
|
||||
by generating step-by-step solutions and providing the correct final answer.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: aime24
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AIME-2024
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- aime24
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 14000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 0.6
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 8
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24/20260313_124640
|
||||
@@ -0,0 +1,358 @@
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24/20260313_124640/configs/task_config_77a80d.yaml
|
||||
2026-03-13 12:46:40 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"aime24"
|
||||
],
|
||||
"dataset_args": {
|
||||
"aime24": {
|
||||
"name": "aime24",
|
||||
"dataset_id": "HuggingFaceH4/aime_2024",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "AIME-2024",
|
||||
"description": "The AIME 2024 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 8,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 14000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 0.6,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24/20260313_124640",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Start loading benchmark dataset: aime24
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Loading dataset HuggingFaceH4/aime_2024 from modelscope > subset: default > split: train ...
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Start evaluating 1 subsets of the aime24: ['default']
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Evaluating subset: default
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Getting predictions for subset: default
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Processing 240 samples, if data is large, it may take a while.
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 12:46:48 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 12:47:48 - evalscope - INFO: Predicting[aime24@default]: 0%| 0/240 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 12:48:00 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=236, inflight=2)
|
||||
2026-03-13 12:48:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=234, inflight=2)
|
||||
2026-03-13 12:48:48 - evalscope - INFO: Predicting[aime24@default]: 1%| 2/240 [Elapsed: 02:00 < Remaining: 3:12:26, 48.52s/it]
|
||||
2026-03-13 12:49:48 - evalscope - INFO: Predicting[aime24@default]: 1%| 2/240 [Elapsed: 03:00 < Remaining: 3:12:26, 48.52s/it]
|
||||
2026-03-13 12:49:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=232, inflight=2)
|
||||
2026-03-13 12:50:02 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=230, inflight=2)
|
||||
2026-03-13 12:50:48 - evalscope - INFO: Predicting[aime24@default]: 2%| 6/240 [Elapsed: 04:00 < Remaining: 1:58:38, 30.42s/it]
|
||||
2026-03-13 12:51:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=228, inflight=2)
|
||||
2026-03-13 12:51:48 - evalscope - INFO: Predicting[aime24@default]: 3%| 8/240 [Elapsed: 05:00 < Remaining: 2:12:31, 34.27s/it]
|
||||
2026-03-13 12:51:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=226, inflight=2)
|
||||
2026-03-13 12:52:48 - evalscope - INFO: Predicting[aime24@default]: 4%| 10/240 [Elapsed: 06:00 < Remaining: 1:47:50, 28.13s/it]
|
||||
2026-03-13 12:53:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=224, inflight=2)
|
||||
2026-03-13 12:53:46 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=222, inflight=2)
|
||||
2026-03-13 12:53:48 - evalscope - INFO: Predicting[aime24@default]: 6%| 14/240 [Elapsed: 07:00 < Remaining: 1:39:44, 26.48s/it]
|
||||
2026-03-13 12:54:18 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=220, inflight=2)
|
||||
2026-03-13 12:54:28 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=218, inflight=2)
|
||||
2026-03-13 12:54:49 - evalscope - INFO: Predicting[aime24@default]: 8%| 18/240 [Elapsed: 08:00 < Remaining: 1:04:08, 17.33s/it]
|
||||
2026-03-13 12:54:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=216, inflight=2)
|
||||
2026-03-13 12:55:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=214, inflight=2)
|
||||
2026-03-13 12:55:49 - evalscope - INFO: Predicting[aime24@default]: 9%| 22/240 [Elapsed: 09:00 < Remaining: 1:00:01, 16.52s/it]
|
||||
2026-03-13 12:56:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=212, inflight=2)
|
||||
2026-03-13 12:56:49 - evalscope - INFO: Predicting[aime24@default]: 10%| 23/240 [Elapsed: 10:00 < Remaining: 1:24:38, 23.40s/it]
|
||||
2026-03-13 12:57:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=210, inflight=2)
|
||||
2026-03-13 12:57:49 - evalscope - INFO: Predicting[aime24@default]: 11%| 26/240 [Elapsed: 11:00 < Remaining: 1:14:22, 20.85s/it]
|
||||
2026-03-13 12:58:15 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=208, inflight=2)
|
||||
2026-03-13 12:58:49 - evalscope - INFO: Predicting[aime24@default]: 12%| 28/240 [Elapsed: 12:00 < Remaining: 1:21:53, 23.18s/it]
|
||||
2026-03-13 12:59:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=206, inflight=2)
|
||||
2026-03-13 12:59:43 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=204, inflight=2)
|
||||
2026-03-13 12:59:49 - evalscope - INFO: Predicting[aime24@default]: 13%| 32/240 [Elapsed: 13:00 < Remaining: 1:16:24, 22.04s/it]
|
||||
2026-03-13 13:00:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=202, inflight=2)
|
||||
2026-03-13 13:00:49 - evalscope - INFO: Predicting[aime24@default]: 14%| 34/240 [Elapsed: 14:00 < Remaining: 1:06:51, 19.47s/it]
|
||||
2026-03-13 13:01:32 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=200, inflight=2)
|
||||
2026-03-13 13:01:49 - evalscope - INFO: Predicting[aime24@default]: 15%| 36/240 [Elapsed: 15:00 < Remaining: 1:27:44, 25.80s/it]
|
||||
2026-03-13 13:02:06 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=198, inflight=2)
|
||||
2026-03-13 13:02:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=196, inflight=2)
|
||||
2026-03-13 13:02:49 - evalscope - INFO: Predicting[aime24@default]: 17%| 40/240 [Elapsed: 16:00 < Remaining: 1:06:23, 19.92s/it]
|
||||
2026-03-13 13:03:49 - evalscope - INFO: Predicting[aime24@default]: 17%| 40/240 [Elapsed: 17:00 < Remaining: 1:06:23, 19.92s/it]
|
||||
2026-03-13 13:04:00 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=194, inflight=2)
|
||||
2026-03-13 13:04:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=192, inflight=2)
|
||||
2026-03-13 13:04:49 - evalscope - INFO: Predicting[aime24@default]: 18%| 44/240 [Elapsed: 18:00 < Remaining: 1:13:13, 22.42s/it]
|
||||
2026-03-13 13:05:49 - evalscope - INFO: Predicting[aime24@default]: 18%| 44/240 [Elapsed: 19:00 < Remaining: 1:13:13, 22.42s/it]
|
||||
2026-03-13 13:05:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=190, inflight=2)
|
||||
2026-03-13 13:06:15 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=188, inflight=2)
|
||||
2026-03-13 13:06:49 - evalscope - INFO: Predicting[aime24@default]: 20%| 48/240 [Elapsed: 20:01 < Remaining: 1:16:21, 23.86s/it]
|
||||
2026-03-13 13:07:17 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=186, inflight=2)
|
||||
2026-03-13 13:07:37 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=184, inflight=2)
|
||||
2026-03-13 13:07:49 - evalscope - INFO: Predicting[aime24@default]: 22%| 52/240 [Elapsed: 21:01 < Remaining: 54:53, 17.52s/it]
|
||||
2026-03-13 13:08:46 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=182, inflight=2)
|
||||
2026-03-13 13:08:49 - evalscope - INFO: Predicting[aime24@default]: 22%| 54/240 [Elapsed: 22:01 < Remaining: 1:25:50, 27.69s/it]
|
||||
2026-03-13 13:09:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=180, inflight=2)
|
||||
2026-03-13 13:09:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=178, inflight=2)
|
||||
2026-03-13 13:09:49 - evalscope - INFO: Predicting[aime24@default]: 24%| 58/240 [Elapsed: 23:01 < Remaining: 54:11, 17.87s/it]
|
||||
2026-03-13 13:10:49 - evalscope - INFO: Predicting[aime24@default]: 24%| 58/240 [Elapsed: 24:01 < Remaining: 54:11, 17.87s/it]
|
||||
2026-03-13 13:10:55 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=176, inflight=2)
|
||||
2026-03-13 13:11:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=174, inflight=2)
|
||||
2026-03-13 13:11:49 - evalscope - INFO: Predicting[aime24@default]: 26%| 62/240 [Elapsed: 25:01 < Remaining: 1:05:11, 21.97s/it]
|
||||
2026-03-13 13:12:24 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=172, inflight=2)
|
||||
2026-03-13 13:12:49 - evalscope - INFO: Predicting[aime24@default]: 27%| 64/240 [Elapsed: 26:01 < Remaining: 1:11:51, 24.50s/it]
|
||||
2026-03-13 13:12:58 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=170, inflight=2)
|
||||
2026-03-13 13:13:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=168, inflight=2)
|
||||
2026-03-13 13:13:45 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=166, inflight=2)
|
||||
2026-03-13 13:13:49 - evalscope - INFO: Predicting[aime24@default]: 29%| 70/240 [Elapsed: 27:01 < Remaining: 47:45, 16.86s/it]
|
||||
2026-03-13 13:14:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=164, inflight=2)
|
||||
2026-03-13 13:14:49 - evalscope - INFO: Predicting[aime24@default]: 30%| 72/240 [Elapsed: 28:01 < Remaining: 43:55, 15.69s/it]
|
||||
2026-03-13 13:15:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=162, inflight=2)
|
||||
2026-03-13 13:15:25 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=160, inflight=2)
|
||||
2026-03-13 13:15:49 - evalscope - INFO: Predicting[aime24@default]: 32%| 76/240 [Elapsed: 29:01 < Remaining: 42:38, 15.60s/it]
|
||||
2026-03-13 13:16:26 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=158, inflight=2)
|
||||
2026-03-13 13:16:49 - evalscope - INFO: Predicting[aime24@default]: 32%| 78/240 [Elapsed: 30:01 < Remaining: 54:14, 20.09s/it]
|
||||
2026-03-13 13:17:13 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=156, inflight=2)
|
||||
2026-03-13 13:17:50 - evalscope - INFO: Predicting[aime24@default]: 33%| 80/240 [Elapsed: 31:01 < Remaining: 56:19, 21.12s/it]
|
||||
2026-03-13 13:18:21 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=154, inflight=2)
|
||||
2026-03-13 13:18:50 - evalscope - INFO: Predicting[aime24@default]: 34%| 82/240 [Elapsed: 32:01 < Remaining: 1:05:49, 25.00s/it]
|
||||
2026-03-13 13:19:01 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=152, inflight=2)
|
||||
2026-03-13 13:19:50 - evalscope - INFO: Predicting[aime24@default]: 35%| 84/240 [Elapsed: 33:01 < Remaining: 1:01:06, 23.50s/it]
|
||||
2026-03-13 13:20:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=150, inflight=2)
|
||||
2026-03-13 13:20:50 - evalscope - INFO: Predicting[aime24@default]: 36%| 86/240 [Elapsed: 34:01 < Remaining: 1:09:57, 27.26s/it]
|
||||
2026-03-13 13:20:54 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=148, inflight=2)
|
||||
2026-03-13 13:21:50 - evalscope - INFO: Predicting[aime24@default]: 37%| 88/240 [Elapsed: 35:01 < Remaining: 1:03:55, 25.23s/it]
|
||||
2026-03-13 13:22:00 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=146, inflight=2)
|
||||
2026-03-13 13:22:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=144, inflight=2)
|
||||
2026-03-13 13:22:50 - evalscope - INFO: Predicting[aime24@default]: 38%| 92/240 [Elapsed: 36:01 < Remaining: 53:09, 21.55s/it]
|
||||
2026-03-13 13:23:34 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=142, inflight=2)
|
||||
2026-03-13 13:23:50 - evalscope - INFO: Predicting[aime24@default]: 39%| 94/240 [Elapsed: 37:01 < Remaining: 1:05:33, 26.94s/it]
|
||||
2026-03-13 13:24:02 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=140, inflight=2)
|
||||
2026-03-13 13:24:42 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=138, inflight=2)
|
||||
2026-03-13 13:24:50 - evalscope - INFO: Predicting[aime24@default]: 41%| 98/240 [Elapsed: 38:01 < Remaining: 52:31, 22.19s/it]
|
||||
2026-03-13 13:25:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=136, inflight=2)
|
||||
2026-03-13 13:25:50 - evalscope - INFO: Predicting[aime24@default]: 42%| 100/240 [Elapsed: 39:01 < Remaining: 50:15, 21.54s/it]
|
||||
2026-03-13 13:25:53 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=134, inflight=2)
|
||||
2026-03-13 13:26:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=132, inflight=2)
|
||||
2026-03-13 13:26:50 - evalscope - INFO: Predicting[aime24@default]: 43%| 104/240 [Elapsed: 40:01 < Remaining: 47:58, 21.16s/it]
|
||||
2026-03-13 13:27:47 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=130, inflight=2)
|
||||
2026-03-13 13:27:50 - evalscope - INFO: Predicting[aime24@default]: 44%| 106/240 [Elapsed: 41:01 < Remaining: 54:52, 24.57s/it]
|
||||
2026-03-13 13:27:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=128, inflight=2)
|
||||
2026-03-13 13:28:50 - evalscope - INFO: Predicting[aime24@default]: 45%| 108/240 [Elapsed: 42:01 < Remaining: 40:28, 18.40s/it]
|
||||
2026-03-13 13:29:37 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=126, inflight=2)
|
||||
2026-03-13 13:29:50 - evalscope - INFO: Predicting[aime24@default]: 46%| 110/240 [Elapsed: 43:01 < Remaining: 1:01:04, 28.19s/it]
|
||||
2026-03-13 13:29:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=124, inflight=2)
|
||||
2026-03-13 13:30:50 - evalscope - INFO: Predicting[aime24@default]: 47%| 112/240 [Elapsed: 44:01 < Remaining: 46:15, 21.68s/it]
|
||||
2026-03-13 13:31:26 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=122, inflight=2)
|
||||
2026-03-13 13:31:42 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=120, inflight=2)
|
||||
2026-03-13 13:31:50 - evalscope - INFO: Predicting[aime24@default]: 48%| 116/240 [Elapsed: 45:01 < Remaining: 47:32, 23.01s/it]
|
||||
2026-03-13 13:32:47 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=118, inflight=2)
|
||||
2026-03-13 13:32:50 - evalscope - INFO: Predicting[aime24@default]: 49%| 118/240 [Elapsed: 46:01 < Remaining: 52:34, 25.86s/it]
|
||||
2026-03-13 13:33:23 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=116, inflight=2)
|
||||
2026-03-13 13:33:50 - evalscope - INFO: Predicting[aime24@default]: 50%| 120/240 [Elapsed: 47:02 < Remaining: 47:18, 23.65s/it]
|
||||
2026-03-13 13:34:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=114, inflight=2)
|
||||
2026-03-13 13:34:50 - evalscope - INFO: Predicting[aime24@default]: 51%| 122/240 [Elapsed: 48:02 < Remaining: 54:42, 27.81s/it]
|
||||
2026-03-13 13:35:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=112, inflight=2)
|
||||
2026-03-13 13:35:50 - evalscope - INFO: Predicting[aime24@default]: 52%| 124/240 [Elapsed: 49:02 < Remaining: 48:22, 25.02s/it]
|
||||
2026-03-13 13:36:32 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=110, inflight=2)
|
||||
2026-03-13 13:36:50 - evalscope - INFO: Predicting[aime24@default]: 52%| 126/240 [Elapsed: 50:02 < Remaining: 55:14, 29.07s/it]
|
||||
2026-03-13 13:37:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=108, inflight=2)
|
||||
2026-03-13 13:37:50 - evalscope - INFO: Predicting[aime24@default]: 53%| 128/240 [Elapsed: 51:02 < Remaining: 48:54, 26.20s/it]
|
||||
2026-03-13 13:38:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=106, inflight=2)
|
||||
2026-03-13 13:38:50 - evalscope - INFO: Predicting[aime24@default]: 54%| 130/240 [Elapsed: 52:02 < Remaining: 54:32, 29.75s/it]
|
||||
2026-03-13 13:39:05 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=104, inflight=2)
|
||||
2026-03-13 13:39:50 - evalscope - INFO: Predicting[aime24@default]: 55%| 132/240 [Elapsed: 53:02 < Remaining: 47:44, 26.53s/it]
|
||||
2026-03-13 13:40:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=102, inflight=2)
|
||||
2026-03-13 13:40:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=100, inflight=2)
|
||||
2026-03-13 13:40:50 - evalscope - INFO: Predicting[aime24@default]: 57%| 136/240 [Elapsed: 54:02 < Remaining: 40:05, 23.13s/it]
|
||||
2026-03-13 13:41:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=98, inflight=2)
|
||||
2026-03-13 13:41:50 - evalscope - INFO: Predicting[aime24@default]: 57%| 137/240 [Elapsed: 55:02 < Remaining: 46:51, 27.29s/it]
|
||||
2026-03-13 13:42:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=96, inflight=2)
|
||||
2026-03-13 13:42:50 - evalscope - INFO: Predicting[aime24@default]: 58%| 140/240 [Elapsed: 56:02 < Remaining: 41:50, 25.11s/it]
|
||||
2026-03-13 13:43:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=94, inflight=2)
|
||||
2026-03-13 13:43:50 - evalscope - INFO: Predicting[aime24@default]: 59%| 142/240 [Elapsed: 57:02 < Remaining: 39:14, 24.03s/it]
|
||||
2026-03-13 13:44:26 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=92, inflight=2)
|
||||
2026-03-13 13:44:49 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=90, inflight=2)
|
||||
2026-03-13 13:44:50 - evalscope - INFO: Predicting[aime24@default]: 61%| 146/240 [Elapsed: 58:02 < Remaining: 35:52, 22.89s/it]
|
||||
2026-03-13 13:45:50 - evalscope - INFO: Predicting[aime24@default]: 61%| 146/240 [Elapsed: 59:02 < Remaining: 35:52, 22.89s/it]
|
||||
2026-03-13 13:46:19 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=88, inflight=2)
|
||||
2026-03-13 13:46:36 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=86, inflight=2)
|
||||
2026-03-13 13:46:50 - evalscope - INFO: Predicting[aime24@default]: 62%| 150/240 [Elapsed: 1:00:02 < Remaining: 35:03, 23.37s/it]
|
||||
2026-03-13 13:47:50 - evalscope - INFO: Predicting[aime24@default]: 62%| 150/240 [Elapsed: 1:01:02 < Remaining: 35:03, 23.37s/it]
|
||||
2026-03-13 13:48:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=84, inflight=2)
|
||||
2026-03-13 13:48:15 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=82, inflight=2)
|
||||
2026-03-13 13:48:51 - evalscope - INFO: Predicting[aime24@default]: 64%| 154/240 [Elapsed: 1:02:02 < Remaining: 31:21, 21.88s/it]
|
||||
2026-03-13 13:49:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=80, inflight=2)
|
||||
2026-03-13 13:49:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=78, inflight=2)
|
||||
2026-03-13 13:49:51 - evalscope - INFO: Predicting[aime24@default]: 66%| 158/240 [Elapsed: 1:03:02 < Remaining: 29:06, 21.30s/it]
|
||||
2026-03-13 13:50:51 - evalscope - INFO: Predicting[aime24@default]: 66%| 158/240 [Elapsed: 1:04:02 < Remaining: 29:06, 21.30s/it]
|
||||
2026-03-13 13:51:21 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=76, inflight=2)
|
||||
2026-03-13 13:51:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=74, inflight=2)
|
||||
2026-03-13 13:51:51 - evalscope - INFO: Predicting[aime24@default]: 68%| 162/240 [Elapsed: 1:05:02 < Remaining: 29:46, 22.91s/it]
|
||||
2026-03-13 13:52:51 - evalscope - INFO: Predicting[aime24@default]: 68%| 162/240 [Elapsed: 1:06:02 < Remaining: 29:46, 22.91s/it]
|
||||
2026-03-13 13:53:14 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=72, inflight=2)
|
||||
2026-03-13 13:53:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=70, inflight=2)
|
||||
2026-03-13 13:53:51 - evalscope - INFO: Predicting[aime24@default]: 69%| 166/240 [Elapsed: 1:07:02 < Remaining: 29:51, 24.21s/it]
|
||||
2026-03-13 13:54:51 - evalscope - INFO: Predicting[aime24@default]: 69%| 166/240 [Elapsed: 1:08:02 < Remaining: 29:51, 24.21s/it]
|
||||
2026-03-13 13:55:01 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=68, inflight=2)
|
||||
2026-03-13 13:55:05 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=66, inflight=2)
|
||||
2026-03-13 13:55:51 - evalscope - INFO: Predicting[aime24@default]: 71%| 170/240 [Elapsed: 1:09:02 < Remaining: 25:11, 21.60s/it]
|
||||
2026-03-13 13:56:51 - evalscope - INFO: Predicting[aime24@default]: 71%| 170/240 [Elapsed: 1:10:02 < Remaining: 25:11, 21.60s/it]
|
||||
2026-03-13 13:56:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=64, inflight=2)
|
||||
2026-03-13 13:56:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=62, inflight=2)
|
||||
2026-03-13 13:57:51 - evalscope - INFO: Predicting[aime24@default]: 72%| 174/240 [Elapsed: 1:11:02 < Remaining: 24:52, 22.62s/it]
|
||||
2026-03-13 13:57:53 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=60, inflight=2)
|
||||
2026-03-13 13:58:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=58, inflight=2)
|
||||
2026-03-13 13:58:51 - evalscope - INFO: Predicting[aime24@default]: 74%| 178/240 [Elapsed: 1:12:02 < Remaining: 24:49, 24.02s/it]
|
||||
2026-03-13 13:59:41 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=56, inflight=2)
|
||||
2026-03-13 13:59:51 - evalscope - INFO: Predicting[aime24@default]: 75%| 180/240 [Elapsed: 1:13:02 < Remaining: 25:57, 25.97s/it]
|
||||
2026-03-13 14:00:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=54, inflight=2)
|
||||
2026-03-13 14:00:51 - evalscope - INFO: Predicting[aime24@default]: 76%| 182/240 [Elapsed: 1:14:02 < Remaining: 25:24, 26.28s/it]
|
||||
2026-03-13 14:01:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=52, inflight=2)
|
||||
2026-03-13 14:01:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=50, inflight=2)
|
||||
2026-03-13 14:01:51 - evalscope - INFO: Predicting[aime24@default]: 78%| 186/240 [Elapsed: 1:15:02 < Remaining: 17:57, 19.96s/it]
|
||||
2026-03-13 14:02:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=48, inflight=2)
|
||||
2026-03-13 14:02:51 - evalscope - INFO: Predicting[aime24@default]: 78%| 188/240 [Elapsed: 1:16:02 < Remaining: 19:54, 22.98s/it]
|
||||
2026-03-13 14:02:54 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=46, inflight=2)
|
||||
2026-03-13 14:03:51 - evalscope - INFO: Predicting[aime24@default]: 79%| 190/240 [Elapsed: 1:17:02 < Remaining: 15:16, 18.33s/it]
|
||||
2026-03-13 14:03:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=44, inflight=2)
|
||||
2026-03-13 14:04:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=42, inflight=2)
|
||||
2026-03-13 14:04:49 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=40, inflight=2)
|
||||
2026-03-13 14:04:51 - evalscope - INFO: Predicting[aime24@default]: 82%| 196/240 [Elapsed: 1:18:02 < Remaining: 12:12, 16.64s/it]
|
||||
2026-03-13 14:05:51 - evalscope - INFO: Predicting[aime24@default]: 82%| 196/240 [Elapsed: 1:19:02 < Remaining: 12:12, 16.64s/it]
|
||||
2026-03-13 14:05:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=38, inflight=2)
|
||||
2026-03-13 14:06:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=36, inflight=2)
|
||||
2026-03-13 14:06:51 - evalscope - INFO: Predicting[aime24@default]: 83%| 200/240 [Elapsed: 1:20:02 < Remaining: 11:52, 17.82s/it]
|
||||
2026-03-13 14:07:47 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=34, inflight=2)
|
||||
2026-03-13 14:07:51 - evalscope - INFO: Predicting[aime24@default]: 84%| 202/240 [Elapsed: 1:21:02 < Remaining: 16:55, 26.73s/it]
|
||||
2026-03-13 14:08:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=32, inflight=2)
|
||||
2026-03-13 14:08:51 - evalscope - INFO: Predicting[aime24@default]: 85%| 204/240 [Elapsed: 1:22:02 < Remaining: 13:01, 21.71s/it]
|
||||
2026-03-13 14:09:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=30, inflight=2)
|
||||
2026-03-13 14:09:51 - evalscope - INFO: Predicting[aime24@default]: 86%| 206/240 [Elapsed: 1:23:03 < Remaining: 16:46, 29.60s/it]
|
||||
2026-03-13 14:10:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=28, inflight=2)
|
||||
2026-03-13 14:10:51 - evalscope - INFO: Predicting[aime24@default]: 87%| 208/240 [Elapsed: 1:24:03 < Remaining: 12:39, 23.72s/it]
|
||||
2026-03-13 14:11:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=26, inflight=2)
|
||||
2026-03-13 14:11:31 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=24, inflight=2)
|
||||
2026-03-13 14:11:51 - evalscope - INFO: Predicting[aime24@default]: 88%| 212/240 [Elapsed: 1:25:03 < Remaining: 09:49, 21.05s/it]
|
||||
2026-03-13 14:12:51 - evalscope - INFO: Predicting[aime24@default]: 88%| 212/240 [Elapsed: 1:26:03 < Remaining: 09:49, 21.05s/it]
|
||||
2026-03-13 14:13:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=22, inflight=2)
|
||||
2026-03-13 14:13:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=20, inflight=2)
|
||||
2026-03-13 14:13:51 - evalscope - INFO: Predicting[aime24@default]: 90%| 216/240 [Elapsed: 1:27:03 < Remaining: 08:46, 21.96s/it]
|
||||
2026-03-13 14:14:51 - evalscope - INFO: Predicting[aime24@default]: 90%| 216/240 [Elapsed: 1:28:03 < Remaining: 08:46, 21.96s/it]
|
||||
2026-03-13 14:15:10 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=18, inflight=2)
|
||||
2026-03-13 14:15:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=16, inflight=2)
|
||||
2026-03-13 14:15:51 - evalscope - INFO: Predicting[aime24@default]: 92%| 220/240 [Elapsed: 1:29:03 < Remaining: 10:34, 31.72s/it]
|
||||
2026-03-13 14:16:51 - evalscope - INFO: Predicting[aime24@default]: 92%| 220/240 [Elapsed: 1:30:03 < Remaining: 10:34, 31.72s/it]
|
||||
2026-03-13 14:16:59 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=14, inflight=2)
|
||||
2026-03-13 14:17:00 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=12, inflight=2)
|
||||
2026-03-13 14:17:51 - evalscope - INFO: Predicting[aime24@default]: 93%| 224/240 [Elapsed: 1:31:03 < Remaining: 05:57, 22.35s/it]
|
||||
2026-03-13 14:18:42 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=10, inflight=2)
|
||||
2026-03-13 14:18:51 - evalscope - INFO: Predicting[aime24@default]: 94%| 226/240 [Elapsed: 1:32:03 < Remaining: 06:59, 29.97s/it]
|
||||
2026-03-13 14:18:55 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=8, inflight=2)
|
||||
2026-03-13 14:19:51 - evalscope - INFO: Predicting[aime24@default]: 95%| 228/240 [Elapsed: 1:33:03 < Remaining: 04:41, 23.45s/it]
|
||||
2026-03-13 14:20:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=6, inflight=2)
|
||||
2026-03-13 14:20:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=4, inflight=2)
|
||||
2026-03-13 14:20:51 - evalscope - INFO: Predicting[aime24@default]: 96%| 231/240 [Elapsed: 1:34:03 < Remaining: 03:38, 24.31s/it]
|
||||
2026-03-13 14:21:10 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=2, inflight=2)
|
||||
2026-03-13 14:21:44 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=0, inflight=2)
|
||||
2026-03-13 14:21:51 - evalscope - INFO: Predicting[aime24@default]: 99%| 238/240 [Elapsed: 1:35:03 < Remaining: 00:28, 14.23s/it]
|
||||
2026-03-13 14:22:51 - evalscope - INFO: Predicting[aime24@default]: 99%| 238/240 [Elapsed: 1:36:03 < Remaining: 00:28, 14.23s/it]
|
||||
2026-03-13 14:23:05 - evalscope - INFO: Predicting[aime24@default]: 100%| 240/240 [Elapsed: 1:36:16 < Remaining: 00:00, 21.21s/it]
|
||||
2026-03-13 14:23:05 - evalscope - INFO: Finished getting predictions for subset: default.
|
||||
2026-03-13 14:23:05 - evalscope - INFO: Getting reviews for subset: default
|
||||
2026-03-13 14:23:05 - evalscope - INFO: Reviewing 240 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Reviewing[aime24@default]: 100%| 240/240 [Elapsed: 00:02 < Remaining: 00:00, 9.61it/s]
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Finished reviewing subset: default. Total reviewed: 240
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Aggregating scores for subset: default
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Evaluating [aime24] 100%| 1/1 [Elapsed: 1:36:19 < Remaining: 00:00, 5779.66s/subset]
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Generating report...
|
||||
2026-03-13 14:23:08 - evalscope - INFO:
|
||||
aime24 report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime24 | mean_acc | default | 240 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24/20260313_124640/reports/llama3_3b_instruct_vallina_full_sft_30k/aime24.json
|
||||
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Benchmark aime24 evaluation finished.
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime24 | mean_acc | default | 240 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['aime24']
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24/20260313_124640
|
||||
2026-03-13 14:23:08 - evalscope - INFO: [进度条] AIME24 评测完成 ✓
|
||||
2026-03-13 14:23:08 - evalscope - INFO: [断点续传] AIME24 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME24_result.json
|
||||
2026-03-13 14:23:08 - evalscope - INFO: 完成评测 AIME24 (2/8)
|
||||
2026-03-13 14:23:08 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-13 14:23:08 - evalscope - INFO: 正在评估 MATH500 (repeat: 1次) (剩余: 5个)
|
||||
2026-03-13 14:23:08 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-13 14:23:08 - evalscope - INFO: 开始创建 benchmark MATH500 的 TaskConfig
|
||||
2026-03-13 14:23:08 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-13 14:23:08 - evalscope - INFO: [MATH500] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['math_500'], eval_batch_size=2048
|
||||
2026-03-13 14:23:08 - evalscope - INFO: 开始评测 MATH500...
|
||||
2026-03-13 14:23:08 - evalscope - INFO: [进度条] MATH500 开始评测
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:012134f365ea7a750e2c5e8457a792da231c510d3939a6d85ad003e65efeeef7
|
||||
size 16438234
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime24",
|
||||
"dataset_name": "aime24",
|
||||
"dataset_pretty_name": "AIME-2024",
|
||||
"dataset_description": "The AIME 2024 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.0167,
|
||||
"num": 240
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
36
evalscope/version_20260313_110435/run_1/AIME24_result.json
Normal file
36
evalscope/version_20260313_110435/run_1/AIME24_result.json
Normal file
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"aime24": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime24",
|
||||
"dataset_name": "aime24",
|
||||
"dataset_pretty_name": "AIME-2024",
|
||||
"dataset_description": "The AIME 2024 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.0167,
|
||||
"num": 240
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
aime25:
|
||||
aggregation: mean
|
||||
dataset_id: opencompass/AIME2025
|
||||
default_subset: default
|
||||
description: The AIME 2025 benchmark is based on problems from the American Invitational
|
||||
Mathematics Examination, a prestigious high school mathematics competition.
|
||||
This benchmark tests a model's ability to solve challenging mathematics problems
|
||||
by generating step-by-step solutions and providing the correct final answer.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: aime25
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AIME-2025
|
||||
prompt_template: '
|
||||
|
||||
Solve the following math problem step by step. Put your answer inside \boxed{{}}.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
Remember to put your answer inside \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- AIME2025-I
|
||||
- AIME2025-II
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- aime25
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 14000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 0.6
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 8
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25/20260313_110445
|
||||
@@ -0,0 +1,385 @@
|
||||
2026-03-13 11:04:45 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 11:04:45 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25/20260313_110445/configs/task_config_7f760c.yaml
|
||||
2026-03-13 11:04:45 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"aime25"
|
||||
],
|
||||
"dataset_args": {
|
||||
"aime25": {
|
||||
"name": "aime25",
|
||||
"dataset_id": "opencompass/AIME2025",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"AIME2025-I",
|
||||
"AIME2025-II"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "test",
|
||||
"prompt_template": "\nSolve the following math problem step by step. Put your answer inside \\boxed{{}}.\n\n{question}\n\nRemember to put your answer inside \\boxed{{}}.",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "AIME-2025",
|
||||
"description": "The AIME 2025 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 8,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 14000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 0.6,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25/20260313_110445",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 11:04:45 - evalscope - INFO: Start loading benchmark dataset: aime25
|
||||
2026-03-13 11:04:49 - evalscope - INFO: Loading dataset opencompass/AIME2025 from modelscope > subset: AIME2025-I > split: test ...
|
||||
2026-03-13 11:04:57 - evalscope - INFO: Loading dataset opencompass/AIME2025 from modelscope > subset: AIME2025-II > split: test ...
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Start evaluating 2 subsets of the aime25: ['AIME2025-I', 'AIME2025-II']
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Evaluating subset: AIME2025-I
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Getting predictions for subset: AIME2025-I
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Processing 120 samples, if data is large, it may take a while.
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 11:05:02 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=115, inflight=2)
|
||||
2026-03-13 11:06:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 0%| 0/120 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 11:07:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 0%| 0/120 [Elapsed: 02:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 11:08:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 0%| 0/120 [Elapsed: 03:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 11:08:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=115, inflight=2)
|
||||
2026-03-13 11:08:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=113, inflight=2)
|
||||
2026-03-13 11:09:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 2%| 3/120 [Elapsed: 04:00 < Remaining: 2:57:52, 91.22s/it]
|
||||
2026-03-13 11:09:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=111, inflight=2)
|
||||
2026-03-13 11:10:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 4%| 5/120 [Elapsed: 05:00 < Remaining: 1:38:53, 51.59s/it]
|
||||
2026-03-13 11:10:29 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=109, inflight=2)
|
||||
2026-03-13 11:11:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 6%| 7/120 [Elapsed: 06:00 < Remaining: 1:14:51, 39.74s/it]
|
||||
2026-03-13 11:11:25 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=107, inflight=2)
|
||||
2026-03-13 11:12:01 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=105, inflight=2)
|
||||
2026-03-13 11:12:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 9%| 11/120 [Elapsed: 07:00 < Remaining: 52:19, 28.80s/it]
|
||||
2026-03-13 11:13:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 9%| 11/120 [Elapsed: 08:00 < Remaining: 52:19, 28.80s/it]
|
||||
2026-03-13 11:13:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=103, inflight=2)
|
||||
2026-03-13 11:13:55 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=101, inflight=2)
|
||||
2026-03-13 11:14:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 12%| 15/120 [Elapsed: 09:00 < Remaining: 48:44, 27.85s/it]
|
||||
2026-03-13 11:14:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=99, inflight=2)
|
||||
2026-03-13 11:15:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 14%| 17/120 [Elapsed: 10:00 < Remaining: 37:44, 21.99s/it]
|
||||
2026-03-13 11:15:42 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=97, inflight=2)
|
||||
2026-03-13 11:16:01 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=95, inflight=2)
|
||||
2026-03-13 11:16:02 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 18%| 21/120 [Elapsed: 11:00 < Remaining: 37:46, 22.90s/it]
|
||||
2026-03-13 11:16:41 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=93, inflight=2)
|
||||
2026-03-13 11:17:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 19%| 23/120 [Elapsed: 12:00 < Remaining: 35:35, 22.01s/it]
|
||||
2026-03-13 11:17:25 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=91, inflight=2)
|
||||
2026-03-13 11:18:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 21%| 25/120 [Elapsed: 13:00 < Remaining: 34:51, 22.01s/it]
|
||||
2026-03-13 11:18:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=89, inflight=2)
|
||||
2026-03-13 11:18:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=87, inflight=2)
|
||||
2026-03-13 11:19:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 24%| 29/120 [Elapsed: 14:00 < Remaining: 27:29, 18.13s/it]
|
||||
2026-03-13 11:19:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=85, inflight=2)
|
||||
2026-03-13 11:20:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 26%| 31/120 [Elapsed: 15:00 < Remaining: 37:07, 25.03s/it]
|
||||
2026-03-13 11:20:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=83, inflight=2)
|
||||
2026-03-13 11:21:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 28%| 33/120 [Elapsed: 16:00 < Remaining: 31:55, 22.01s/it]
|
||||
2026-03-13 11:21:37 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=81, inflight=2)
|
||||
2026-03-13 11:21:45 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=79, inflight=2)
|
||||
2026-03-13 11:22:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 31%| 37/120 [Elapsed: 17:00 < Remaining: 27:28, 19.86s/it]
|
||||
2026-03-13 11:22:16 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=77, inflight=2)
|
||||
2026-03-13 11:23:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 32%| 39/120 [Elapsed: 18:00 < Remaining: 25:02, 18.55s/it]
|
||||
2026-03-13 11:23:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=75, inflight=2)
|
||||
2026-03-13 11:24:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 34%| 41/120 [Elapsed: 19:00 < Remaining: 33:30, 25.45s/it]
|
||||
2026-03-13 11:24:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=73, inflight=2)
|
||||
2026-03-13 11:24:44 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=71, inflight=2)
|
||||
2026-03-13 11:25:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 38%| 45/120 [Elapsed: 20:00 < Remaining: 26:25, 21.14s/it]
|
||||
2026-03-13 11:25:23 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=69, inflight=2)
|
||||
2026-03-13 11:26:00 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=67, inflight=2)
|
||||
2026-03-13 11:26:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 41%| 49/120 [Elapsed: 21:00 < Remaining: 23:40, 20.01s/it]
|
||||
2026-03-13 11:26:31 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=65, inflight=2)
|
||||
2026-03-13 11:27:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 42%| 51/120 [Elapsed: 22:00 < Remaining: 21:27, 18.66s/it]
|
||||
2026-03-13 11:27:47 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=63, inflight=2)
|
||||
2026-03-13 11:28:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 44%| 53/120 [Elapsed: 23:00 < Remaining: 27:19, 24.47s/it]
|
||||
2026-03-13 11:28:19 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=61, inflight=2)
|
||||
2026-03-13 11:29:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 46%| 55/120 [Elapsed: 24:00 < Remaining: 23:45, 21.93s/it]
|
||||
2026-03-13 11:29:29 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=59, inflight=2)
|
||||
2026-03-13 11:29:39 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=57, inflight=2)
|
||||
2026-03-13 11:30:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 49%| 59/120 [Elapsed: 25:00 < Remaining: 19:58, 19.64s/it]
|
||||
2026-03-13 11:30:26 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=55, inflight=2)
|
||||
2026-03-13 11:31:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 51%| 61/120 [Elapsed: 26:00 < Remaining: 20:27, 20.80s/it]
|
||||
2026-03-13 11:31:04 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=53, inflight=2)
|
||||
2026-03-13 11:32:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 52%| 63/120 [Elapsed: 27:00 < Remaining: 19:06, 20.11s/it]
|
||||
2026-03-13 11:32:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=51, inflight=2)
|
||||
2026-03-13 11:32:57 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=49, inflight=2)
|
||||
2026-03-13 11:33:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 56%| 67/120 [Elapsed: 28:00 < Remaining: 21:01, 23.81s/it]
|
||||
2026-03-13 11:34:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 56%| 67/120 [Elapsed: 29:00 < Remaining: 21:01, 23.81s/it]
|
||||
2026-03-13 11:34:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=47, inflight=2)
|
||||
2026-03-13 11:34:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=45, inflight=2)
|
||||
2026-03-13 11:35:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 59%| 71/120 [Elapsed: 30:00 < Remaining: 21:11, 25.95s/it]
|
||||
2026-03-13 11:35:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=43, inflight=2)
|
||||
2026-03-13 11:35:58 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=41, inflight=2)
|
||||
2026-03-13 11:36:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 62%| 75/120 [Elapsed: 31:00 < Remaining: 14:56, 19.92s/it]
|
||||
2026-03-13 11:37:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 62%| 75/120 [Elapsed: 32:00 < Remaining: 14:56, 19.92s/it]
|
||||
2026-03-13 11:37:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 11:37:51 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 11:38:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 66%| 79/120 [Elapsed: 33:00 < Remaining: 14:54, 21.81s/it]
|
||||
2026-03-13 11:38:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 11:39:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 68%| 81/120 [Elapsed: 34:00 < Remaining: 15:28, 23.82s/it]
|
||||
2026-03-13 11:39:44 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 11:40:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 69%| 83/120 [Elapsed: 35:00 < Remaining: 15:27, 25.08s/it]
|
||||
2026-03-13 11:40:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 11:41:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 71%| 85/120 [Elapsed: 36:00 < Remaining: 15:08, 25.96s/it]
|
||||
2026-03-13 11:41:09 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 11:42:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 72%| 87/120 [Elapsed: 37:00 < Remaining: 12:18, 22.37s/it]
|
||||
2026-03-13 11:42:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 11:43:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 74%| 89/120 [Elapsed: 38:00 < Remaining: 14:45, 28.56s/it]
|
||||
2026-03-13 11:43:04 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 11:44:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 76%| 91/120 [Elapsed: 39:00 < Remaining: 11:46, 24.35s/it]
|
||||
2026-03-13 11:44:25 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 11:44:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 11:45:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 79%| 95/120 [Elapsed: 40:01 < Remaining: 10:37, 25.49s/it]
|
||||
2026-03-13 11:46:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 79%| 95/120 [Elapsed: 41:01 < Remaining: 10:37, 25.49s/it]
|
||||
2026-03-13 11:46:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 11:46:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 11:47:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 82%| 99/120 [Elapsed: 42:01 < Remaining: 08:57, 25.59s/it]
|
||||
2026-03-13 11:47:58 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 11:48:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 84%| 101/120 [Elapsed: 43:01 < Remaining: 08:57, 28.27s/it]
|
||||
2026-03-13 11:48:16 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 11:49:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 86%| 103/120 [Elapsed: 44:01 < Remaining: 06:19, 22.34s/it]
|
||||
2026-03-13 11:49:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 11:50:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 11:50:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 88%| 106/120 [Elapsed: 45:01 < Remaining: 05:21, 22.95s/it]
|
||||
2026-03-13 11:51:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 89%| 107/120 [Elapsed: 46:01 < Remaining: 04:58, 22.95s/it]
|
||||
2026-03-13 11:51:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 11:51:58 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 11:52:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 92%| 111/120 [Elapsed: 47:01 < Remaining: 03:40, 24.45s/it]
|
||||
2026-03-13 11:53:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 92%| 111/120 [Elapsed: 48:01 < Remaining: 03:40, 24.45s/it]
|
||||
2026-03-13 11:53:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 11:53:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 11:54:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 96%| 115/120 [Elapsed: 49:01 < Remaining: 01:50, 22.05s/it]
|
||||
2026-03-13 11:55:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 96%| 115/120 [Elapsed: 50:01 < Remaining: 01:50, 22.05s/it]
|
||||
2026-03-13 11:55:13 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 11:56:03 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 99%| 119/120 [Elapsed: 51:01 < Remaining: 00:22, 22.37s/it]
|
||||
2026-03-13 11:56:21 - evalscope - INFO: Predicting[aime25@AIME2025-I]: 100%| 120/120 [Elapsed: 51:18 < Remaining: 00:00, 25.07s/it]
|
||||
2026-03-13 11:56:21 - evalscope - INFO: Finished getting predictions for subset: AIME2025-I.
|
||||
2026-03-13 11:56:21 - evalscope - INFO: Getting reviews for subset: AIME2025-I
|
||||
2026-03-13 11:56:21 - evalscope - INFO: Reviewing 120 samples, if data is large, it may take a while.
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Reviewing[aime25@AIME2025-I]: 100%| 120/120 [Elapsed: 00:02 < Remaining: 00:00, 2.19s/it]
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Finished reviewing subset: AIME2025-I. Total reviewed: 120
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Aggregating scores for subset: AIME2025-I
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Evaluating [aime25] 50%| 1/2 [Elapsed: 51:21 < Remaining: 51:21, 3081.37s/subset]
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Evaluating subset: AIME2025-II
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Getting predictions for subset: AIME2025-II
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Processing 120 samples, if data is large, it may take a while.
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 11:56:23 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=114, inflight=2)
|
||||
2026-03-13 11:57:23 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=115, inflight=2)
|
||||
2026-03-13 11:57:23 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 1%| 1/120 [Elapsed: 01:00 < Remaining: 1:59:03, 60.03s/it]
|
||||
2026-03-13 11:57:45 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=113, inflight=2)
|
||||
2026-03-13 11:58:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 2%| 3/120 [Elapsed: 02:00 < Remaining: 1:13:27, 37.67s/it]
|
||||
2026-03-13 11:59:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=111, inflight=2)
|
||||
2026-03-13 11:59:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 4%| 5/120 [Elapsed: 03:00 < Remaining: 1:20:25, 41.96s/it]
|
||||
2026-03-13 11:59:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=109, inflight=2)
|
||||
2026-03-13 12:00:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 6%| 7/120 [Elapsed: 04:00 < Remaining: 53:16, 28.28s/it]
|
||||
2026-03-13 12:00:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=107, inflight=2)
|
||||
2026-03-13 12:01:09 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=105, inflight=2)
|
||||
2026-03-13 12:01:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 9%| 11/120 [Elapsed: 05:00 < Remaining: 43:08, 23.75s/it]
|
||||
2026-03-13 12:01:34 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=103, inflight=2)
|
||||
2026-03-13 12:02:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 11%| 13/120 [Elapsed: 06:00 < Remaining: 35:28, 19.89s/it]
|
||||
2026-03-13 12:02:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=101, inflight=2)
|
||||
2026-03-13 12:03:02 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=99, inflight=2)
|
||||
2026-03-13 12:03:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 14%| 17/120 [Elapsed: 07:00 < Remaining: 34:58, 20.37s/it]
|
||||
2026-03-13 12:03:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=97, inflight=2)
|
||||
2026-03-13 12:04:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=95, inflight=2)
|
||||
2026-03-13 12:04:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 18%| 21/120 [Elapsed: 08:00 < Remaining: 32:09, 19.49s/it]
|
||||
2026-03-13 12:05:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 18%| 21/120 [Elapsed: 09:00 < Remaining: 32:09, 19.49s/it]
|
||||
2026-03-13 12:05:46 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=93, inflight=2)
|
||||
2026-03-13 12:06:02 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=91, inflight=2)
|
||||
2026-03-13 12:06:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 21%| 25/120 [Elapsed: 10:00 < Remaining: 32:55, 20.79s/it]
|
||||
2026-03-13 12:07:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=89, inflight=2)
|
||||
2026-03-13 12:07:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 22%| 27/120 [Elapsed: 11:00 < Remaining: 40:47, 26.32s/it]
|
||||
2026-03-13 12:07:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=87, inflight=2)
|
||||
2026-03-13 12:08:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 24%| 29/120 [Elapsed: 12:00 < Remaining: 32:40, 21.54s/it]
|
||||
2026-03-13 12:09:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=85, inflight=2)
|
||||
2026-03-13 12:09:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 26%| 31/120 [Elapsed: 13:00 < Remaining: 41:46, 28.17s/it]
|
||||
2026-03-13 12:09:37 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=83, inflight=2)
|
||||
2026-03-13 12:10:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 28%| 33/120 [Elapsed: 14:00 < Remaining: 34:39, 23.90s/it]
|
||||
2026-03-13 12:11:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=81, inflight=2)
|
||||
2026-03-13 12:11:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 29%| 35/120 [Elapsed: 15:00 < Remaining: 42:13, 29.80s/it]
|
||||
2026-03-13 12:11:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=79, inflight=2)
|
||||
2026-03-13 12:12:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 31%| 37/120 [Elapsed: 16:00 < Remaining: 34:52, 25.21s/it]
|
||||
2026-03-13 12:12:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=77, inflight=2)
|
||||
2026-03-13 12:13:06 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=75, inflight=2)
|
||||
2026-03-13 12:13:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 34%| 41/120 [Elapsed: 17:00 < Remaining: 29:40, 22.54s/it]
|
||||
2026-03-13 12:14:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=73, inflight=2)
|
||||
2026-03-13 12:14:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 36%| 43/120 [Elapsed: 18:00 < Remaining: 34:30, 26.88s/it]
|
||||
2026-03-13 12:14:52 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=71, inflight=2)
|
||||
2026-03-13 12:15:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 38%| 45/120 [Elapsed: 19:00 < Remaining: 29:20, 23.47s/it]
|
||||
2026-03-13 12:16:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=69, inflight=2)
|
||||
2026-03-13 12:16:17 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=67, inflight=2)
|
||||
2026-03-13 12:16:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 41%| 49/120 [Elapsed: 20:00 < Remaining: 34:32, 29.19s/it]
|
||||
2026-03-13 12:17:01 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=65, inflight=2)
|
||||
2026-03-13 12:17:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 42%| 51/120 [Elapsed: 21:00 < Remaining: 23:54, 20.79s/it]
|
||||
2026-03-13 12:18:05 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=63, inflight=2)
|
||||
2026-03-13 12:18:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 44%| 53/120 [Elapsed: 22:00 < Remaining: 26:19, 23.58s/it]
|
||||
2026-03-13 12:18:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=61, inflight=2)
|
||||
2026-03-13 12:19:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 46%| 55/120 [Elapsed: 23:00 < Remaining: 24:57, 23.04s/it]
|
||||
2026-03-13 12:19:29 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=59, inflight=2)
|
||||
2026-03-13 12:20:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 48%| 57/120 [Elapsed: 24:00 < Remaining: 23:36, 22.48s/it]
|
||||
2026-03-13 12:20:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=57, inflight=2)
|
||||
2026-03-13 12:21:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=55, inflight=2)
|
||||
2026-03-13 12:21:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 50%| 60/120 [Elapsed: 25:00 < Remaining: 24:51, 24.85s/it]
|
||||
2026-03-13 12:22:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 51%| 61/120 [Elapsed: 26:00 < Remaining: 24:26, 24.85s/it]
|
||||
2026-03-13 12:22:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=53, inflight=2)
|
||||
2026-03-13 12:22:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=51, inflight=2)
|
||||
2026-03-13 12:23:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 54%| 65/120 [Elapsed: 27:00 < Remaining: 18:16, 19.94s/it]
|
||||
2026-03-13 12:23:38 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=49, inflight=2)
|
||||
2026-03-13 12:24:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=47, inflight=2)
|
||||
2026-03-13 12:24:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 57%| 69/120 [Elapsed: 28:00 < Remaining: 19:15, 22.66s/it]
|
||||
2026-03-13 12:25:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 57%| 69/120 [Elapsed: 29:00 < Remaining: 19:15, 22.66s/it]
|
||||
2026-03-13 12:25:34 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=45, inflight=2)
|
||||
2026-03-13 12:26:07 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=43, inflight=2)
|
||||
2026-03-13 12:26:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 61%| 73/120 [Elapsed: 30:00 < Remaining: 18:39, 23.82s/it]
|
||||
2026-03-13 12:27:23 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=41, inflight=2)
|
||||
2026-03-13 12:27:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 62%| 75/120 [Elapsed: 31:00 < Remaining: 21:02, 28.06s/it]
|
||||
2026-03-13 12:27:54 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 12:28:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 64%| 77/120 [Elapsed: 32:00 < Remaining: 17:25, 24.30s/it]
|
||||
2026-03-13 12:28:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 12:29:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 12:29:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 68%| 81/120 [Elapsed: 33:00 < Remaining: 13:57, 21.49s/it]
|
||||
2026-03-13 12:30:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 12:30:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 69%| 83/120 [Elapsed: 34:00 < Remaining: 14:55, 24.19s/it]
|
||||
2026-03-13 12:30:33 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 12:31:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 71%| 85/120 [Elapsed: 35:00 < Remaining: 11:43, 20.09s/it]
|
||||
2026-03-13 12:31:53 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 12:32:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 12:32:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 74%| 89/120 [Elapsed: 36:00 < Remaining: 11:40, 22.60s/it]
|
||||
2026-03-13 12:33:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 74%| 89/120 [Elapsed: 37:00 < Remaining: 11:40, 22.60s/it]
|
||||
2026-03-13 12:33:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 12:34:17 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 12:34:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 78%| 93/120 [Elapsed: 38:00 < Remaining: 11:02, 24.55s/it]
|
||||
2026-03-13 12:35:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 78%| 93/120 [Elapsed: 39:01 < Remaining: 11:02, 24.55s/it]
|
||||
2026-03-13 12:35:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 12:36:12 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 12:36:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 81%| 97/120 [Elapsed: 40:01 < Remaining: 09:57, 26.00s/it]
|
||||
2026-03-13 12:37:24 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 81%| 97/120 [Elapsed: 41:01 < Remaining: 09:57, 26.00s/it]
|
||||
2026-03-13 12:37:26 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 12:38:06 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 12:38:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 84%| 101/120 [Elapsed: 42:01 < Remaining: 08:23, 26.51s/it]
|
||||
2026-03-13 12:39:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 12:39:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 86%| 103/120 [Elapsed: 43:01 < Remaining: 08:26, 29.81s/it]
|
||||
2026-03-13 12:39:45 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 12:40:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 88%| 105/120 [Elapsed: 44:01 < Remaining: 06:07, 24.47s/it]
|
||||
2026-03-13 12:41:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 12:41:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 89%| 107/120 [Elapsed: 45:01 < Remaining: 06:40, 30.78s/it]
|
||||
2026-03-13 12:41:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 12:42:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 91%| 109/120 [Elapsed: 46:01 < Remaining: 04:36, 25.15s/it]
|
||||
2026-03-13 12:43:10 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 12:43:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 92%| 111/120 [Elapsed: 47:01 < Remaining: 04:39, 31.11s/it]
|
||||
2026-03-13 12:43:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 12:44:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 94%| 113/120 [Elapsed: 48:01 < Remaining: 02:54, 24.93s/it]
|
||||
2026-03-13 12:44:38 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 12:45:22 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 12:45:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 98%| 117/120 [Elapsed: 49:01 < Remaining: 01:17, 25.85s/it]
|
||||
2026-03-13 12:46:25 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 98%| 118/120 [Elapsed: 50:01 < Remaining: 00:49, 25.00s/it]
|
||||
2026-03-13 12:46:33 - evalscope - INFO: Predicting[aime25@AIME2025-II]: 100%| 120/120 [Elapsed: 50:09 < Remaining: 00:00, 24.82s/it]
|
||||
2026-03-13 12:46:33 - evalscope - INFO: Finished getting predictions for subset: AIME2025-II.
|
||||
2026-03-13 12:46:33 - evalscope - INFO: Getting reviews for subset: AIME2025-II
|
||||
2026-03-13 12:46:33 - evalscope - INFO: Reviewing 120 samples, if data is large, it may take a while.
|
||||
2026-03-13 12:46:38 - evalscope - INFO: Reviewing[aime25@AIME2025-II]: 100%| 120/120 [Elapsed: 00:05 < Remaining: 00:00, 5.03s/it]
|
||||
2026-03-13 12:46:38 - evalscope - INFO: Finished reviewing subset: AIME2025-II. Total reviewed: 120
|
||||
2026-03-13 12:46:38 - evalscope - INFO: Aggregating scores for subset: AIME2025-II
|
||||
2026-03-13 12:46:38 - evalscope - INFO: Evaluating [aime25] 100%| 2/2 [Elapsed: 1:41:35 < Remaining: 00:00, 3042.00s/subset]
|
||||
2026-03-13 12:46:38 - evalscope - INFO: Generating report...
|
||||
2026-03-13 12:46:39 - evalscope - INFO:
|
||||
aime25 report table:
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+=============+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | AIME2025-I | 120 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | AIME2025-II | 120 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | OVERALL | 240 | 0.0167 | - |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
|
||||
2026-03-13 12:46:39 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-13 12:46:39 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25/20260313_110445/reports/llama3_3b_instruct_vallina_full_sft_30k/aime25.json
|
||||
|
||||
2026-03-13 12:46:39 - evalscope - INFO: Benchmark aime25 evaluation finished.
|
||||
2026-03-13 12:46:39 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+=============+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | AIME2025-I | 120 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | AIME2025-II | 120 | 0.0167 | default |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | aime25 | mean_acc | OVERALL | 240 | 0.0167 | - |
|
||||
+-----------------------------------------+-----------+----------+-------------+-------+---------+---------+
|
||||
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['aime25']
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25/20260313_110445
|
||||
2026-03-13 12:46:40 - evalscope - INFO: [进度条] AIME25 评测完成 ✓
|
||||
2026-03-13 12:46:40 - evalscope - INFO: [断点续传] AIME25 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AIME25_result.json
|
||||
2026-03-13 12:46:40 - evalscope - INFO: 完成评测 AIME25 (1/8)
|
||||
2026-03-13 12:46:40 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-13 12:46:40 - evalscope - INFO: 正在评估 AIME24 (repeat: 8次) (剩余: 6个)
|
||||
2026-03-13 12:46:40 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-13 12:46:40 - evalscope - INFO: 开始创建 benchmark AIME24 的 TaskConfig
|
||||
2026-03-13 12:46:40 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-13 12:46:40 - evalscope - INFO: [AIME24] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['aime24'], eval_batch_size=2048
|
||||
2026-03-13 12:46:40 - evalscope - INFO: 开始评测 AIME24...
|
||||
2026-03-13 12:46:40 - evalscope - INFO: [进度条] AIME24 开始评测
|
||||
2026-03-13 12:46:40 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime25",
|
||||
"dataset_name": "aime25",
|
||||
"dataset_pretty_name": "AIME-2025",
|
||||
"dataset_description": "The AIME 2025 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "AIME2025-I",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
},
|
||||
{
|
||||
"name": "AIME2025-II",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
41
evalscope/version_20260313_110435/run_1/AIME25_result.json
Normal file
41
evalscope/version_20260313_110435/run_1/AIME25_result.json
Normal file
@@ -0,0 +1,41 @@
|
||||
{
|
||||
"aime25": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime25",
|
||||
"dataset_name": "aime25",
|
||||
"dataset_pretty_name": "AIME-2025",
|
||||
"dataset_description": "The AIME 2025 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "AIME2025-I",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
},
|
||||
{
|
||||
"name": "AIME2025-II",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,78 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
amc:
|
||||
aggregation: mean
|
||||
dataset_id: evalscope/amc_22-24
|
||||
default_subset: default
|
||||
description: AMC (American Mathematics Competitions) is a series of mathematics
|
||||
competitions for high school students.
|
||||
eval_split: null
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: amc
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AMC
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- amc22
|
||||
- amc23
|
||||
- amc24
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- amc
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC/20260313_164421
|
||||
@@ -0,0 +1,293 @@
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC/20260313_164421/configs/task_config_eefae5.yaml
|
||||
2026-03-13 16:44:21 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"amc"
|
||||
],
|
||||
"dataset_args": {
|
||||
"amc": {
|
||||
"name": "amc",
|
||||
"dataset_id": "evalscope/amc_22-24",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"amc22",
|
||||
"amc23",
|
||||
"amc24"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": null,
|
||||
"prompt_template": "{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "AMC",
|
||||
"description": "AMC (American Mathematics Competitions) is a series of mathematics competitions for high school students.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC/20260313_164421",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Start loading benchmark dataset: amc
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Loading dataset evalscope/amc_22-24 from modelscope > subset: default > split: amc22 ...
|
||||
2026-03-13 16:44:33 - evalscope - INFO: Loading dataset evalscope/amc_22-24 from modelscope > subset: default > split: amc23 ...
|
||||
2026-03-13 16:44:37 - evalscope - INFO: Loading dataset evalscope/amc_22-24 from modelscope > subset: default > split: amc24 ...
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Start evaluating 3 subsets of the amc: ['amc22', 'amc23', 'amc24']
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Evaluating subset: amc22
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Getting predictions for subset: amc22
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Processing 43 samples, if data is large, it may take a while.
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 16:44:41 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=40, inflight=2)
|
||||
2026-03-13 16:45:14 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=38, inflight=2)
|
||||
2026-03-13 16:45:41 - evalscope - INFO: Predicting[amc@amc22]: 5%| 2/43 [Elapsed: 01:00 < Remaining: 22:33, 33.01s/it]
|
||||
2026-03-13 16:46:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=36, inflight=2)
|
||||
2026-03-13 16:46:41 - evalscope - INFO: Predicting[amc@amc22]: 7%| 3/43 [Elapsed: 02:00 < Remaining: 19:47, 29.68s/it]
|
||||
2026-03-13 16:46:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=34, inflight=2)
|
||||
2026-03-13 16:47:41 - evalscope - INFO: Predicting[amc@amc22]: 12%| 5/43 [Elapsed: 03:00 < Remaining: 20:24, 32.22s/it]
|
||||
2026-03-13 16:47:43 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=32, inflight=2)
|
||||
2026-03-13 16:47:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=30, inflight=2)
|
||||
2026-03-13 16:48:41 - evalscope - INFO: Predicting[amc@amc22]: 21%| 9/43 [Elapsed: 04:00 < Remaining: 16:55, 29.88s/it]
|
||||
2026-03-13 16:49:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=28, inflight=2)
|
||||
2026-03-13 16:49:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=26, inflight=2)
|
||||
2026-03-13 16:49:41 - evalscope - INFO: Predicting[amc@amc22]: 30%| 13/43 [Elapsed: 05:00 < Remaining: 09:26, 18.88s/it]
|
||||
2026-03-13 16:50:29 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=24, inflight=2)
|
||||
2026-03-13 16:50:41 - evalscope - INFO: Predicting[amc@amc22]: 35%| 15/43 [Elapsed: 06:00 < Remaining: 10:54, 23.37s/it]
|
||||
2026-03-13 16:50:53 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=22, inflight=2)
|
||||
2026-03-13 16:51:41 - evalscope - INFO: Predicting[amc@amc22]: 40%| 17/43 [Elapsed: 07:00 < Remaining: 08:39, 19.99s/it]
|
||||
2026-03-13 16:51:58 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=20, inflight=2)
|
||||
2026-03-13 16:52:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=18, inflight=2)
|
||||
2026-03-13 16:52:41 - evalscope - INFO: Predicting[amc@amc22]: 49%| 21/43 [Elapsed: 08:00 < Remaining: 06:21, 17.34s/it]
|
||||
2026-03-13 16:53:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=16, inflight=2)
|
||||
2026-03-13 16:53:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=14, inflight=2)
|
||||
2026-03-13 16:53:41 - evalscope - INFO: Predicting[amc@amc22]: 58%| 25/43 [Elapsed: 09:00 < Remaining: 05:54, 19.67s/it]
|
||||
2026-03-13 16:54:25 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=12, inflight=2)
|
||||
2026-03-13 16:54:41 - evalscope - INFO: Predicting[amc@amc22]: 63%| 27/43 [Elapsed: 10:00 < Remaining: 05:37, 21.12s/it]
|
||||
2026-03-13 16:55:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=10, inflight=2)
|
||||
2026-03-13 16:55:24 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=8, inflight=2)
|
||||
2026-03-13 16:55:41 - evalscope - INFO: Predicting[amc@amc22]: 72%| 31/43 [Elapsed: 11:00 < Remaining: 03:22, 16.89s/it]
|
||||
2026-03-13 16:56:41 - evalscope - INFO: Predicting[amc@amc22]: 72%| 31/43 [Elapsed: 12:00 < Remaining: 03:22, 16.89s/it]
|
||||
2026-03-13 16:56:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=6, inflight=2)
|
||||
2026-03-13 16:57:01 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=4, inflight=2)
|
||||
2026-03-13 16:57:41 - evalscope - INFO: Predicting[amc@amc22]: 81%| 35/43 [Elapsed: 13:00 < Remaining: 02:30, 18.87s/it]
|
||||
2026-03-13 16:57:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=2, inflight=2)
|
||||
2026-03-13 16:58:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=0, inflight=2)
|
||||
2026-03-13 16:58:41 - evalscope - INFO: Predicting[amc@amc22]: 91%| 39/43 [Elapsed: 14:00 < Remaining: 01:16, 19.22s/it]
|
||||
2026-03-13 16:59:41 - evalscope - INFO: Predicting[amc@amc22]: 95%| 41/43 [Elapsed: 15:00 < Remaining: 00:47, 23.81s/it]
|
||||
2026-03-13 16:59:45 - evalscope - INFO: Predicting[amc@amc22]: 100%| 43/43 [Elapsed: 15:04 < Remaining: 00:00, 18.47s/it]
|
||||
2026-03-13 16:59:45 - evalscope - INFO: Finished getting predictions for subset: amc22.
|
||||
2026-03-13 16:59:45 - evalscope - INFO: Getting reviews for subset: amc22
|
||||
2026-03-13 16:59:45 - evalscope - INFO: Reviewing 43 samples, if data is large, it may take a while.
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Reviewing[amc@amc22]: 100%| 43/43 [Elapsed: 00:00 < Remaining: 00:00, 74.05it/s]
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Finished reviewing subset: amc22. Total reviewed: 43
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Aggregating scores for subset: amc22
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Evaluating [amc] 33%| 1/3 [Elapsed: 15:04 < Remaining: 30:09, 904.97s/subset]
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Evaluating subset: amc23
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Getting predictions for subset: amc23
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Processing 46 samples, if data is large, it may take a while.
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 16:59:46 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 16:59:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=42, inflight=2)
|
||||
2026-03-13 17:00:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=40, inflight=2)
|
||||
2026-03-13 17:00:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=38, inflight=2)
|
||||
2026-03-13 17:00:46 - evalscope - INFO: Predicting[amc@amc23]: 9%| 4/46 [Elapsed: 01:00 < Remaining: 08:24, 12.01s/it]
|
||||
2026-03-13 17:01:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=36, inflight=2)
|
||||
2026-03-13 17:01:46 - evalscope - INFO: Predicting[amc@amc23]: 13%| 6/46 [Elapsed: 02:00 < Remaining: 14:23, 21.60s/it]
|
||||
2026-03-13 17:02:12 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=34, inflight=2)
|
||||
2026-03-13 17:02:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=32, inflight=2)
|
||||
2026-03-13 17:02:46 - evalscope - INFO: Predicting[amc@amc23]: 22%| 10/46 [Elapsed: 03:00 < Remaining: 10:47, 17.98s/it]
|
||||
2026-03-13 17:03:46 - evalscope - INFO: Predicting[amc@amc23]: 22%| 10/46 [Elapsed: 04:00 < Remaining: 10:47, 17.98s/it]
|
||||
2026-03-13 17:03:49 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=30, inflight=2)
|
||||
2026-03-13 17:03:53 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=28, inflight=2)
|
||||
2026-03-13 17:04:46 - evalscope - INFO: Predicting[amc@amc23]: 30%| 14/46 [Elapsed: 05:00 < Remaining: 09:21, 17.55s/it]
|
||||
2026-03-13 17:05:21 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=26, inflight=2)
|
||||
2026-03-13 17:05:31 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=24, inflight=2)
|
||||
2026-03-13 17:05:46 - evalscope - INFO: Predicting[amc@amc23]: 39%| 18/46 [Elapsed: 06:00 < Remaining: 09:04, 19.46s/it]
|
||||
2026-03-13 17:06:46 - evalscope - INFO: Predicting[amc@amc23]: 39%| 18/46 [Elapsed: 07:00 < Remaining: 09:04, 19.46s/it]
|
||||
2026-03-13 17:06:54 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=22, inflight=2)
|
||||
2026-03-13 17:07:01 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=20, inflight=2)
|
||||
2026-03-13 17:07:15 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=18, inflight=2)
|
||||
2026-03-13 17:07:46 - evalscope - INFO: Predicting[amc@amc23]: 52%| 24/46 [Elapsed: 08:00 < Remaining: 05:42, 15.58s/it]
|
||||
2026-03-13 17:07:53 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=16, inflight=2)
|
||||
2026-03-13 17:08:18 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=14, inflight=2)
|
||||
2026-03-13 17:08:46 - evalscope - INFO: Predicting[amc@amc23]: 61%| 28/46 [Elapsed: 09:00 < Remaining: 04:37, 15.42s/it]
|
||||
2026-03-13 17:09:02 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=12, inflight=2)
|
||||
2026-03-13 17:09:46 - evalscope - INFO: Predicting[amc@amc23]: 65%| 30/46 [Elapsed: 10:00 < Remaining: 04:38, 17.41s/it]
|
||||
2026-03-13 17:09:54 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=10, inflight=2)
|
||||
2026-03-13 17:10:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=8, inflight=2)
|
||||
2026-03-13 17:10:46 - evalscope - INFO: Predicting[amc@amc23]: 74%| 34/46 [Elapsed: 11:00 < Remaining: 04:07, 20.65s/it]
|
||||
2026-03-13 17:10:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=6, inflight=2)
|
||||
2026-03-13 17:11:46 - evalscope - INFO: Predicting[amc@amc23]: 78%| 36/46 [Elapsed: 12:00 < Remaining: 02:49, 17.00s/it]
|
||||
2026-03-13 17:12:06 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=4, inflight=2)
|
||||
2026-03-13 17:12:15 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=2, inflight=2)
|
||||
2026-03-13 17:12:46 - evalscope - INFO: Predicting[amc@amc23]: 87%| 40/46 [Elapsed: 13:00 < Remaining: 01:42, 17.03s/it]
|
||||
2026-03-13 17:13:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=0, inflight=2)
|
||||
2026-03-13 17:13:46 - evalscope - INFO: Predicting[amc@amc23]: 96%| 44/46 [Elapsed: 14:00 < Remaining: 00:36, 18.02s/it]
|
||||
2026-03-13 17:14:46 - evalscope - INFO: Predicting[amc@amc23]: 96%| 44/46 [Elapsed: 15:00 < Remaining: 00:36, 18.02s/it]
|
||||
2026-03-13 17:15:08 - evalscope - INFO: Predicting[amc@amc23]: 100%| 46/46 [Elapsed: 15:22 < Remaining: 00:00, 25.50s/it]
|
||||
2026-03-13 17:15:08 - evalscope - INFO: Finished getting predictions for subset: amc23.
|
||||
2026-03-13 17:15:08 - evalscope - INFO: Getting reviews for subset: amc23
|
||||
2026-03-13 17:15:08 - evalscope - INFO: Reviewing 46 samples, if data is large, it may take a while.
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Reviewing[amc@amc23]: 100%| 46/46 [Elapsed: 00:00 < Remaining: 00:00, 99.32it/s]
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Finished reviewing subset: amc23. Total reviewed: 46
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Aggregating scores for subset: amc23
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Evaluating [amc] 67%| 2/3 [Elapsed: 30:27 < Remaining: 15:15, 915.38s/subset]
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Evaluating subset: amc24
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Getting predictions for subset: amc24
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Processing 45 samples, if data is large, it may take a while.
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 17:15:09 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 17:15:32 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=41, inflight=2)
|
||||
2026-03-13 17:16:09 - evalscope - INFO: Predicting[amc@amc24]: 2%| 1/45 [Elapsed: 01:00 < Remaining: 16:52, 23.01s/it]
|
||||
2026-03-13 17:16:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 17:17:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 17:17:09 - evalscope - INFO: Predicting[amc@amc24]: 9%| 4/45 [Elapsed: 02:00 < Remaining: 26:42, 39.09s/it]
|
||||
2026-03-13 17:18:02 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 17:18:09 - evalscope - INFO: Predicting[amc@amc24]: 13%| 6/45 [Elapsed: 03:00 < Remaining: 21:52, 33.66s/it]
|
||||
2026-03-13 17:18:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 17:19:09 - evalscope - INFO: Predicting[amc@amc24]: 18%| 8/45 [Elapsed: 04:00 < Remaining: 14:55, 24.20s/it]
|
||||
2026-03-13 17:19:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 17:20:04 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 17:20:09 - evalscope - INFO: Predicting[amc@amc24]: 27%| 12/45 [Elapsed: 05:00 < Remaining: 12:47, 23.24s/it]
|
||||
2026-03-13 17:21:09 - evalscope - INFO: Predicting[amc@amc24]: 27%| 12/45 [Elapsed: 06:00 < Remaining: 12:47, 23.24s/it]
|
||||
2026-03-13 17:21:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 17:21:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 17:22:09 - evalscope - INFO: Predicting[amc@amc24]: 36%| 16/45 [Elapsed: 07:00 < Remaining: 09:29, 19.64s/it]
|
||||
2026-03-13 17:22:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 17:22:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 17:23:09 - evalscope - INFO: Predicting[amc@amc24]: 44%| 20/45 [Elapsed: 08:00 < Remaining: 08:01, 19.26s/it]
|
||||
2026-03-13 17:23:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 17:24:07 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 17:24:09 - evalscope - INFO: Predicting[amc@amc24]: 53%| 24/45 [Elapsed: 09:00 < Remaining: 06:49, 19.48s/it]
|
||||
2026-03-13 17:24:23 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 17:25:09 - evalscope - INFO: Predicting[amc@amc24]: 58%| 26/45 [Elapsed: 10:00 < Remaining: 05:00, 15.84s/it]
|
||||
2026-03-13 17:25:39 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 17:25:58 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 17:26:09 - evalscope - INFO: Predicting[amc@amc24]: 67%| 30/45 [Elapsed: 11:00 < Remaining: 04:41, 18.76s/it]
|
||||
2026-03-13 17:27:09 - evalscope - INFO: Predicting[amc@amc24]: 67%| 30/45 [Elapsed: 12:00 < Remaining: 04:41, 18.76s/it]
|
||||
2026-03-13 17:27:12 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 17:27:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 17:28:09 - evalscope - INFO: Predicting[amc@amc24]: 76%| 34/45 [Elapsed: 13:00 < Remaining: 03:41, 20.12s/it]
|
||||
2026-03-13 17:28:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 17:29:04 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 17:29:09 - evalscope - INFO: Predicting[amc@amc24]: 84%| 38/45 [Elapsed: 14:00 < Remaining: 02:30, 21.44s/it]
|
||||
2026-03-13 17:29:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 17:30:09 - evalscope - INFO: Predicting[amc@amc24]: 89%| 40/45 [Elapsed: 15:00 < Remaining: 01:53, 22.66s/it]
|
||||
2026-03-13 17:30:39 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 17:31:09 - evalscope - INFO: Predicting[amc@amc24]: 93%| 42/45 [Elapsed: 16:00 < Remaining: 01:06, 22.32s/it]
|
||||
2026-03-13 17:31:33 - evalscope - INFO: Predicting[amc@amc24]: 100%| 45/45 [Elapsed: 16:24 < Remaining: 00:00, 19.67s/it]
|
||||
2026-03-13 17:31:33 - evalscope - INFO: Finished getting predictions for subset: amc24.
|
||||
2026-03-13 17:31:33 - evalscope - INFO: Getting reviews for subset: amc24
|
||||
2026-03-13 17:31:33 - evalscope - INFO: Reviewing 45 samples, if data is large, it may take a while.
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Reviewing[amc@amc24]: 100%| 45/45 [Elapsed: 00:00 < Remaining: 00:00, 51.70it/s]
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Finished reviewing subset: amc24. Total reviewed: 45
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Aggregating scores for subset: amc24
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Evaluating [amc] 100%| 3/3 [Elapsed: 46:52 < Remaining: 00:00, 947.20s/subset]
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Generating report...
|
||||
2026-03-13 17:31:34 - evalscope - INFO:
|
||||
amc report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc22 | 43 | 0.1163 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc23 | 46 | 0.2391 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc24 | 45 | 0.1333 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | OVERALL | 134 | 0.1642 | - |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC/20260313_164421/reports/llama3_3b_instruct_vallina_full_sft_30k/amc.json
|
||||
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Benchmark amc evaluation finished.
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc22 | 43 | 0.1163 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc23 | 46 | 0.2391 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | amc24 | 45 | 0.1333 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | amc | mean_acc | OVERALL | 134 | 0.1642 | - |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['amc']
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC/20260313_164421
|
||||
2026-03-13 17:31:34 - evalscope - INFO: [进度条] AMC 评测完成 ✓
|
||||
2026-03-13 17:31:34 - evalscope - INFO: [断点续传] AMC 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/AMC_result.json
|
||||
2026-03-13 17:31:34 - evalscope - INFO: 完成评测 AMC (4/8)
|
||||
2026-03-13 17:31:34 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-13 17:31:34 - evalscope - INFO: 正在评估 AGIEvalMath (repeat: 1次) (剩余: 3个)
|
||||
2026-03-13 17:31:34 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-13 17:31:34 - evalscope - INFO: 开始创建 benchmark AGIEvalMath 的 TaskConfig
|
||||
2026-03-13 17:31:34 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-13 17:31:34 - evalscope - INFO: [AGIEvalMath] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['agieval_math'], eval_batch_size=2048
|
||||
2026-03-13 17:31:34 - evalscope - INFO: 开始评测 AGIEvalMath...
|
||||
2026-03-13 17:31:34 - evalscope - INFO: [进度条] AGIEvalMath 开始评测
|
||||
2026-03-13 17:31:34 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,44 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@amc",
|
||||
"dataset_name": "amc",
|
||||
"dataset_pretty_name": "AMC",
|
||||
"dataset_description": "AMC (American Mathematics Competitions) is a series of mathematics competitions for high school students.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.1642,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1642,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1629,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "amc22",
|
||||
"score": 0.1163,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "amc23",
|
||||
"score": 0.2391,
|
||||
"num": 46
|
||||
},
|
||||
{
|
||||
"name": "amc24",
|
||||
"score": 0.1333,
|
||||
"num": 45
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
46
evalscope/version_20260313_110435/run_1/AMC_result.json
Normal file
46
evalscope/version_20260313_110435/run_1/AMC_result.json
Normal file
@@ -0,0 +1,46 @@
|
||||
{
|
||||
"amc": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@amc",
|
||||
"dataset_name": "amc",
|
||||
"dataset_pretty_name": "AMC",
|
||||
"dataset_description": "AMC (American Mathematics Competitions) is a series of mathematics competitions for high school students.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.1642,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1642,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1629,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "amc22",
|
||||
"score": 0.1163,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "amc23",
|
||||
"score": 0.2391,
|
||||
"num": 46
|
||||
},
|
||||
{
|
||||
"name": "amc24",
|
||||
"score": 0.1333,
|
||||
"num": 45
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
arc:
|
||||
aggregation: mean
|
||||
dataset_id: allenai/ai2_arc
|
||||
default_subset: default
|
||||
description: 'The ARC (AI2 Reasoning Challenge) benchmark is designed to evaluate
|
||||
the reasoning capabilities of AI models through multiple-choice questions derived
|
||||
from science exams. It includes two subsets: ARC-Easy and ARC-Challenge, which
|
||||
vary in difficulty.'
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: arc
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: ARC
|
||||
prompt_template: 'Answer the following multiple choice question. The entire content
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- ARC-Easy
|
||||
- ARC-Challenge
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Reasoning
|
||||
- MCQ
|
||||
train_split: train
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- arc
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC/20260314_194124
|
||||
@@ -0,0 +1,255 @@
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC/20260314_194124/configs/task_config_c8b2f5.yaml
|
||||
2026-03-14 19:41:24 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"arc"
|
||||
],
|
||||
"dataset_args": {
|
||||
"arc": {
|
||||
"name": "arc",
|
||||
"dataset_id": "allenai/ai2_arc",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"ARC-Easy",
|
||||
"ARC-Challenge"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": "train",
|
||||
"eval_split": "test",
|
||||
"prompt_template": "Answer the following multiple choice question. The entire content of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "ARC",
|
||||
"description": "The ARC (AI2 Reasoning Challenge) benchmark is designed to evaluate the reasoning capabilities of AI models through multiple-choice questions derived from science exams. It includes two subsets: ARC-Easy and ARC-Challenge, which vary in difficulty.",
|
||||
"tags": [
|
||||
"Reasoning",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC/20260314_194124",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Start loading benchmark dataset: arc
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Loading dataset allenai/ai2_arc from modelscope > subset: ARC-Easy > split: test ...
|
||||
2026-03-14 19:41:32 - evalscope - INFO: Loading dataset allenai/ai2_arc from modelscope > subset: ARC-Challenge > split: test ...
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Start evaluating 2 subsets of the arc: ['ARC-Easy', 'ARC-Challenge']
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Evaluating subset: ARC-Easy
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Getting predictions for subset: ARC-Easy
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Processing 2376 samples, if data is large, it may take a while.
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:41:40 - evalscope - INFO: Dispatcher: Worker-1 <- 123 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:41:44 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=1796, inflight=2)
|
||||
2026-03-14 19:42:10 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=1669, inflight=2)
|
||||
2026-03-14 19:42:25 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=1669, inflight=2)
|
||||
2026-03-14 19:42:42 - evalscope - INFO: Predicting[arc@ARC-Easy]: 11%| 253/2376 [Elapsed: 01:00 < Remaining: 09:32, 3.71it/s]
|
||||
2026-03-14 19:42:43 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=1664, inflight=2)
|
||||
2026-03-14 19:43:14 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=1611, inflight=2)
|
||||
2026-03-14 19:43:19 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=1483, inflight=2)
|
||||
2026-03-14 19:43:43 - evalscope - INFO: Predicting[arc@ARC-Easy]: 27%| 637/2376 [Elapsed: 02:01 < Remaining: 04:11, 6.93it/s]
|
||||
2026-03-14 19:43:45 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=1355, inflight=2)
|
||||
2026-03-14 19:43:53 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=1227, inflight=2)
|
||||
2026-03-14 19:44:17 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=1099, inflight=2)
|
||||
2026-03-14 19:44:40 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=971, inflight=2)
|
||||
2026-03-14 19:44:44 - evalscope - INFO: Predicting[arc@ARC-Easy]: 48%| 1149/2376 [Elapsed: 03:02 < Remaining: 03:13, 6.33it/s]
|
||||
2026-03-14 19:44:59 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=843, inflight=2)
|
||||
2026-03-14 19:45:06 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=715, inflight=2)
|
||||
2026-03-14 19:45:15 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=587, inflight=2)
|
||||
2026-03-14 19:45:25 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=459, inflight=2)
|
||||
2026-03-14 19:45:45 - evalscope - INFO: Predicting[arc@ARC-Easy]: 70%| 1661/2376 [Elapsed: 04:03 < Remaining: 01:10, 10.12it/s]
|
||||
2026-03-14 19:45:48 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=331, inflight=2)
|
||||
2026-03-14 19:45:55 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=203, inflight=2)
|
||||
2026-03-14 19:46:18 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=75, inflight=2)
|
||||
2026-03-14 19:46:32 - evalscope - INFO: Dispatcher: Worker-0 <- 75 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:46:45 - evalscope - INFO: Predicting[arc@ARC-Easy]: 91%| 2173/2376 [Elapsed: 05:03 < Remaining: 00:24, 8.20it/s]
|
||||
2026-03-14 19:47:31 - evalscope - INFO: Predicting[arc@ARC-Easy]: 100%| 2376/2376 [Elapsed: 05:49 < Remaining: 00:00, 4.85it/s]
|
||||
2026-03-14 19:47:31 - evalscope - INFO: Finished getting predictions for subset: ARC-Easy.
|
||||
2026-03-14 19:47:31 - evalscope - INFO: Getting reviews for subset: ARC-Easy
|
||||
2026-03-14 19:47:31 - evalscope - INFO: Reviewing 2376 samples, if data is large, it may take a while.
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Reviewing[arc@ARC-Easy]: 100%| 2376/2376 [Elapsed: 00:02 < Remaining: 00:00, 1009.07it/s]
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Finished reviewing subset: ARC-Easy. Total reviewed: 2376
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Aggregating scores for subset: ARC-Easy
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Evaluating [arc] 50%| 1/2 [Elapsed: 05:52 < Remaining: 05:52, 353.00s/subset]
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Evaluating subset: ARC-Challenge
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Getting predictions for subset: ARC-Challenge
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Processing 1172 samples, if data is large, it may take a while.
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-14 19:47:33 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:47:34 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=1042, inflight=2)
|
||||
2026-03-14 19:48:31 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=914, inflight=2)
|
||||
2026-03-14 19:48:34 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 0%| 2/1172 [Elapsed: 01:00 < Remaining: 11:04:28, 34.08s/it]
|
||||
2026-03-14 19:48:44 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=786, inflight=2)
|
||||
2026-03-14 19:49:34 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 11%| 130/1172 [Elapsed: 02:00 < Remaining: 6:57:06, 24.02s/it]
|
||||
2026-03-14 19:49:59 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=658, inflight=2)
|
||||
2026-03-14 19:50:15 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=530, inflight=2)
|
||||
2026-03-14 19:50:35 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 33%| 386/1172 [Elapsed: 03:01 < Remaining: 05:40, 2.31it/s]
|
||||
2026-03-14 19:50:56 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=402, inflight=2)
|
||||
2026-03-14 19:51:35 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=274, inflight=2)
|
||||
2026-03-14 19:51:35 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 44%| 515/1172 [Elapsed: 04:01 < Remaining: 03:49, 2.86it/s]
|
||||
2026-03-14 19:52:35 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 55%| 642/1172 [Elapsed: 05:01 < Remaining: 03:05, 2.86it/s]
|
||||
2026-03-14 19:52:37 - evalscope - INFO: Dispatcher: Worker-1 <- 128 prompts (pending=146, inflight=2)
|
||||
2026-03-14 19:52:39 - evalscope - INFO: Dispatcher: Worker-0 <- 128 prompts (pending=18, inflight=2)
|
||||
2026-03-14 19:53:08 - evalscope - INFO: Dispatcher: Worker-0 <- 18 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:53:36 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 98%| 1154/1172 [Elapsed: 06:02 < Remaining: 00:04, 4.41it/s]
|
||||
2026-03-14 19:54:09 - evalscope - INFO: Predicting[arc@ARC-Challenge]: 100%| 1172/1172 [Elapsed: 06:34 < Remaining: 00:00, 4.00it/s]
|
||||
2026-03-14 19:54:09 - evalscope - INFO: Finished getting predictions for subset: ARC-Challenge.
|
||||
2026-03-14 19:54:09 - evalscope - INFO: Getting reviews for subset: ARC-Challenge
|
||||
2026-03-14 19:54:09 - evalscope - INFO: Reviewing 1172 samples, if data is large, it may take a while.
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Reviewing[arc@ARC-Challenge]: 100%| 1172/1172 [Elapsed: 00:01 < Remaining: 00:00, 993.58it/s]
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Finished reviewing subset: ARC-Challenge. Total reviewed: 1172
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Aggregating scores for subset: ARC-Challenge
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Evaluating [arc] 100%| 2/2 [Elapsed: 12:29 < Remaining: 00:00, 378.70s/subset]
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Generating report...
|
||||
2026-03-14 19:54:10 - evalscope - INFO:
|
||||
arc report table:
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+===============+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | ARC-Easy | 2376 | 0.8274 | default |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | ARC-Challenge | 1172 | 0.715 | default |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | OVERALL | 3548 | 0.7903 | - |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC/20260314_194124/reports/llama3_3b_instruct_vallina_full_sft_30k/arc.json
|
||||
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Benchmark arc evaluation finished.
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+===============+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | ARC-Easy | 2376 | 0.8274 | default |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | ARC-Challenge | 1172 | 0.715 | default |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | arc | mean_acc | OVERALL | 3548 | 0.7903 | - |
|
||||
+-----------------------------------------+-----------+----------+---------------+-------+---------+---------+
|
||||
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['arc']
|
||||
2026-03-14 19:54:10 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC/20260314_194124
|
||||
2026-03-14 19:54:10 - evalscope - INFO: [进度条] ARC 评测完成 ✓
|
||||
2026-03-14 19:54:10 - evalscope - INFO: [断点续传] ARC 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/ARC_result.json
|
||||
2026-03-14 19:54:10 - evalscope - INFO: 完成评测 ARC (9/9)
|
||||
2026-03-14 19:54:12 - evalscope - INFO: Excel汇总结果已保存至: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/summary.xlsx
|
||||
2026-03-14 19:54:12 - evalscope - INFO: 详细结果已保存至: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/detailed_results.json
|
||||
2026-03-14 19:54:12 - evalscope - INFO: 第 1 次运行完成
|
||||
2026-03-14 19:54:12 - evalscope - INFO: 等待 5 秒后开始下一次运行...
|
||||
2026-03-14 19:54:17 - evalscope - INFO:
|
||||
######################################################################
|
||||
2026-03-14 19:54:17 - evalscope - INFO: # 第 2/3 次运行
|
||||
2026-03-14 19:54:17 - evalscope - INFO: ######################################################################
|
||||
|
||||
2026-03-14 19:54:17 - evalscope - INFO:
|
||||
============================================================
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始运行评估 (第 2 次运行) - 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 配置文件: eval_configs/llama3_3b_instruct_vallina_full_sft_30k.yaml
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 输出目录: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_2
|
||||
2026-03-14 19:54:17 - evalscope - INFO: ============================================================
|
||||
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 配置已保存至: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_2/config.yaml
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark AIME25 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME25] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME25] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['aime25'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark AIME24 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME24] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME24] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['aime24'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark MATH500 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [MATH500] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [MATH500] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['math_500'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark AMC 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AMC] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AMC] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['amc'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark AGIEvalMath 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AGIEvalMath] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AGIEvalMath] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['agieval_math'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark GSM8K 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [GSM8K] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [GSM8K] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['gsm8k'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark GPQA 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [GPQA] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [GPQA] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['gpqa_extend'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark MMLUProNoMath 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [MMLUProNoMath] worker_chunk_size=256
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [MMLUProNoMath] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['mmlu_pro_no_math'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark ARC 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [ARC] worker_chunk_size=128
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [ARC] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['arc'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 有效 benchmarks: ['AIME25', 'AIME24', 'MATH500', 'AMC', 'AGIEvalMath', 'GSM8K', 'GPQA', 'MMLUProNoMath', 'ARC']
|
||||
2026-03-14 19:54:17 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 正在评估 AIME25 (repeat: 8次) (剩余: 8个)
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始创建 benchmark AIME25 的 TaskConfig
|
||||
2026-03-14 19:54:17 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME25] worker_chunk_size=8
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [AIME25] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['aime25'], eval_batch_size=2048
|
||||
2026-03-14 19:54:17 - evalscope - INFO: 开始评测 AIME25...
|
||||
2026-03-14 19:54:17 - evalscope - INFO: [进度条] AIME25 开始评测
|
||||
2026-03-14 19:54:17 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,39 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@arc",
|
||||
"dataset_name": "arc",
|
||||
"dataset_pretty_name": "ARC",
|
||||
"dataset_description": "The ARC (AI2 Reasoning Challenge) benchmark is designed to evaluate the reasoning capabilities of AI models through multiple-choice questions derived from science exams. It includes two subsets: ARC-Easy and ARC-Challenge, which vary in difficulty.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7903,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7903,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7712,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "ARC-Easy",
|
||||
"score": 0.8274,
|
||||
"num": 2376
|
||||
},
|
||||
{
|
||||
"name": "ARC-Challenge",
|
||||
"score": 0.715,
|
||||
"num": 1172
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
41
evalscope/version_20260313_110435/run_1/ARC_result.json
Normal file
41
evalscope/version_20260313_110435/run_1/ARC_result.json
Normal file
@@ -0,0 +1,41 @@
|
||||
{
|
||||
"arc": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@arc",
|
||||
"dataset_name": "arc",
|
||||
"dataset_pretty_name": "ARC",
|
||||
"dataset_description": "The ARC (AI2 Reasoning Challenge) benchmark is designed to evaluate the reasoning capabilities of AI models through multiple-choice questions derived from science exams. It includes two subsets: ARC-Easy and ARC-Challenge, which vary in difficulty.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7903,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7903,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7712,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "ARC-Easy",
|
||||
"score": 0.8274,
|
||||
"num": 2376
|
||||
},
|
||||
{
|
||||
"name": "ARC-Challenge",
|
||||
"score": 0.715,
|
||||
"num": 1172
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_002426
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_002426/configs/task_config_244c70.yaml
|
||||
2026-03-14 00:24:26 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_002426",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 00:24:32 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 00:24:32 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 00:24:33 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103543
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 10:35:43 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 10:35:43 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103543/configs/task_config_b3fb25.yaml
|
||||
2026-03-14 10:35:43 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103543",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 10:35:43 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 10:35:45 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 10:35:48 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 10:35:48 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 10:35:50 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103732
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 10:37:32 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 10:37:32 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103732/configs/task_config_e5e1aa.yaml
|
||||
2026-03-14 10:37:32 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103732",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 10:37:32 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 10:37:33 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 10:37:37 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 10:37:37 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 10:37:39 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103911
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 10:39:11 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 10:39:11 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103911/configs/task_config_31eeba.yaml
|
||||
2026-03-14 10:39:11 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_103911",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 10:39:11 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 10:39:12 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 10:39:15 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 10:39:15 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 10:39:17 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_104635
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 10:46:35 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 10:46:35 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_104635/configs/task_config_13e6bd.yaml
|
||||
2026-03-14 10:46:35 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_104635",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 10:46:35 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 10:46:36 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 10:46:39 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 10:46:39 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 10:46:41 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: default
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_105755
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 10:57:55 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 10:57:55 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_105755/configs/task_config_8262cf.yaml
|
||||
2026-03-14 10:57:55 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_105755",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 10:57:55 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 10:57:55 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 10:57:59 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 10:57:59 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 10:58:00 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: gpqa
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110133
|
||||
@@ -0,0 +1,91 @@
|
||||
2026-03-14 11:01:33 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 11:01:33 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110133/configs/task_config_4f714a.yaml
|
||||
2026-03-14 11:01:33 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "gpqa",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110133",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 11:01:33 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 11:01:33 - evalscope - INFO: Loading dataset modelscope/gpqa from modelscope > subset: default > split: train ...
|
||||
2026-03-14 11:01:37 - evalscope - ERROR: [进度条] GPQA 评测失败 ✗
|
||||
2026-03-14 11:01:37 - evalscope - ERROR: 评测过程中出错: 'default'
|
||||
2026-03-14 11:01:39 - evalscope - INFO: Ray Workers 已关闭
|
||||
@@ -0,0 +1,82 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gpqa_extend:
|
||||
aggregation: mean
|
||||
dataset_id: modelscope/gpqa
|
||||
default_subset: gpqa
|
||||
description: GPQA Extended dataset for evaluating reasoning on graduate-level
|
||||
science problems.
|
||||
eval_split: train
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: gpqa_extend
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GPQA-Extended
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
|
||||
{choices}'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Knowledge
|
||||
- MCQ
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gpqa_extend
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 5
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110837
|
||||
@@ -0,0 +1,268 @@
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110837/configs/task_config_a0fa77.yaml
|
||||
2026-03-14 11:08:37 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gpqa_extend"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gpqa_extend": {
|
||||
"name": "gpqa_extend",
|
||||
"dataset_id": "modelscope/gpqa",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "gpqa",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "train",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\n{question}\n\n{choices}",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GPQA-Extended",
|
||||
"description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"tags": [
|
||||
"Knowledge",
|
||||
"MCQ"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110837",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 5,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox"
|
||||
},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Start loading benchmark dataset: gpqa_extend
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Start evaluating 1 subsets of the gpqa_extend: ['gpqa']
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Evaluating subset: gpqa
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Getting predictions for subset: gpqa
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Processing 546 samples, if data is large, it may take a while.
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-14 11:08:37 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-14 11:08:38 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=2, inflight=1)
|
||||
2026-03-14 11:08:38 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=1, inflight=2)
|
||||
2026-03-14 11:09:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 0%| 0/546 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-14 11:10:09 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=536, inflight=2)
|
||||
2026-03-14 11:10:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 0%| 1/546 [Elapsed: 02:00 < Remaining: 13:56:57, 92.14s/it]
|
||||
2026-03-14 11:11:01 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=528, inflight=2)
|
||||
2026-03-14 11:11:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 0%| 2/546 [Elapsed: 03:00 < Remaining: 10:21:45, 68.58s/it]
|
||||
2026-03-14 11:11:57 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=520, inflight=2)
|
||||
2026-03-14 11:12:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 2%| 10/546 [Elapsed: 04:00 < Remaining: 9:21:39, 62.87s/it]
|
||||
2026-03-14 11:12:50 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=512, inflight=2)
|
||||
2026-03-14 11:13:19 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=504, inflight=2)
|
||||
2026-03-14 11:13:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 5%| 26/546 [Elapsed: 05:00 < Remaining: 1:18:32, 9.06s/it]
|
||||
2026-03-14 11:14:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 5%| 26/546 [Elapsed: 06:00 < Remaining: 1:18:32, 9.06s/it]
|
||||
2026-03-14 11:14:44 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=496, inflight=2)
|
||||
2026-03-14 11:14:55 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=488, inflight=2)
|
||||
2026-03-14 11:15:38 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 8%| 42/546 [Elapsed: 07:00 < Remaining: 54:51, 6.53s/it]
|
||||
2026-03-14 11:16:28 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=480, inflight=2)
|
||||
2026-03-14 11:16:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 9%| 50/546 [Elapsed: 08:00 < Remaining: 1:08:58, 8.34s/it]
|
||||
2026-03-14 11:16:49 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=472, inflight=2)
|
||||
2026-03-14 11:17:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 11%| 58/546 [Elapsed: 09:00 < Remaining: 52:14, 6.42s/it]
|
||||
2026-03-14 11:18:09 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=464, inflight=2)
|
||||
2026-03-14 11:18:31 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=456, inflight=2)
|
||||
2026-03-14 11:18:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 14%| 74/546 [Elapsed: 10:00 < Remaining: 47:39, 6.06s/it]
|
||||
2026-03-14 11:19:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 14%| 74/546 [Elapsed: 11:01 < Remaining: 47:39, 6.06s/it]
|
||||
2026-03-14 11:19:57 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=448, inflight=2)
|
||||
2026-03-14 11:20:09 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=440, inflight=2)
|
||||
2026-03-14 11:20:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 16%| 90/546 [Elapsed: 12:01 < Remaining: 43:05, 5.67s/it]
|
||||
2026-03-14 11:21:24 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=432, inflight=2)
|
||||
2026-03-14 11:21:27 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=424, inflight=2)
|
||||
2026-03-14 11:21:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 19%| 106/546 [Elapsed: 13:01 < Remaining: 35:18, 4.81s/it]
|
||||
2026-03-14 11:22:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 19%| 106/546 [Elapsed: 14:01 < Remaining: 35:18, 4.81s/it]
|
||||
2026-03-14 11:22:40 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=416, inflight=2)
|
||||
2026-03-14 11:23:29 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=408, inflight=2)
|
||||
2026-03-14 11:23:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 22%| 122/546 [Elapsed: 15:01 < Remaining: 43:17, 6.13s/it]
|
||||
2026-03-14 11:24:29 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=400, inflight=2)
|
||||
2026-03-14 11:24:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 24%| 130/546 [Elapsed: 16:01 < Remaining: 45:22, 6.54s/it]
|
||||
2026-03-14 11:25:27 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=392, inflight=2)
|
||||
2026-03-14 11:25:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 25%| 138/546 [Elapsed: 17:01 < Remaining: 45:57, 6.76s/it]
|
||||
2026-03-14 11:25:47 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=384, inflight=2)
|
||||
2026-03-14 11:26:39 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 27%| 146/546 [Elapsed: 18:01 < Remaining: 36:32, 5.48s/it]
|
||||
2026-03-14 11:27:10 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=376, inflight=2)
|
||||
2026-03-14 11:27:34 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=368, inflight=2)
|
||||
2026-03-14 11:27:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 30%| 162/546 [Elapsed: 19:01 < Remaining: 36:55, 5.77s/it]
|
||||
2026-03-14 11:28:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 30%| 162/546 [Elapsed: 20:01 < Remaining: 36:55, 5.77s/it]
|
||||
2026-03-14 11:29:01 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=360, inflight=2)
|
||||
2026-03-14 11:29:23 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=352, inflight=2)
|
||||
2026-03-14 11:29:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 33%| 178/546 [Elapsed: 21:01 < Remaining: 36:25, 5.94s/it]
|
||||
2026-03-14 11:30:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 33%| 178/546 [Elapsed: 22:01 < Remaining: 36:25, 5.94s/it]
|
||||
2026-03-14 11:30:59 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=344, inflight=2)
|
||||
2026-03-14 11:31:10 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=336, inflight=2)
|
||||
2026-03-14 11:31:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 36%| 194/546 [Elapsed: 23:02 < Remaining: 34:18, 5.85s/it]
|
||||
2026-03-14 11:32:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 36%| 194/546 [Elapsed: 24:02 < Remaining: 34:18, 5.85s/it]
|
||||
2026-03-14 11:32:47 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=328, inflight=2)
|
||||
2026-03-14 11:33:02 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=320, inflight=2)
|
||||
2026-03-14 11:33:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 38%| 210/546 [Elapsed: 25:02 < Remaining: 33:32, 5.99s/it]
|
||||
2026-03-14 11:34:14 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=312, inflight=2)
|
||||
2026-03-14 11:34:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 40%| 218/546 [Elapsed: 26:02 < Remaining: 37:29, 6.86s/it]
|
||||
2026-03-14 11:34:47 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=304, inflight=2)
|
||||
2026-03-14 11:35:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 41%| 226/546 [Elapsed: 27:02 < Remaining: 32:24, 6.08s/it]
|
||||
2026-03-14 11:35:57 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=296, inflight=2)
|
||||
2026-03-14 11:36:38 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=288, inflight=2)
|
||||
2026-03-14 11:36:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 44%| 242/546 [Elapsed: 28:02 < Remaining: 32:04, 6.33s/it]
|
||||
2026-03-14 11:37:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 44%| 242/546 [Elapsed: 29:02 < Remaining: 32:04, 6.33s/it]
|
||||
2026-03-14 11:37:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=280, inflight=2)
|
||||
2026-03-14 11:38:33 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=272, inflight=2)
|
||||
2026-03-14 11:38:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 47%| 258/546 [Elapsed: 30:02 < Remaining: 31:49, 6.63s/it]
|
||||
2026-03-14 11:39:34 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=264, inflight=2)
|
||||
2026-03-14 11:39:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 49%| 266/546 [Elapsed: 31:02 < Remaining: 32:20, 6.93s/it]
|
||||
2026-03-14 11:40:19 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=256, inflight=2)
|
||||
2026-03-14 11:40:40 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 50%| 274/546 [Elapsed: 32:02 < Remaining: 29:39, 6.54s/it]
|
||||
2026-03-14 11:41:19 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=248, inflight=2)
|
||||
2026-03-14 11:41:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 52%| 282/546 [Elapsed: 33:02 < Remaining: 30:03, 6.83s/it]
|
||||
2026-03-14 11:42:17 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=240, inflight=2)
|
||||
2026-03-14 11:42:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 53%| 290/546 [Elapsed: 34:02 < Remaining: 29:41, 6.96s/it]
|
||||
2026-03-14 11:43:09 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=232, inflight=2)
|
||||
2026-03-14 11:43:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 55%| 298/546 [Elapsed: 35:02 < Remaining: 28:12, 6.82s/it]
|
||||
2026-03-14 11:44:08 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=224, inflight=2)
|
||||
2026-03-14 11:44:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 56%| 306/546 [Elapsed: 36:02 < Remaining: 27:57, 6.99s/it]
|
||||
2026-03-14 11:45:02 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=216, inflight=2)
|
||||
2026-03-14 11:45:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 58%| 314/546 [Elapsed: 37:03 < Remaining: 26:45, 6.92s/it]
|
||||
2026-03-14 11:45:57 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=208, inflight=2)
|
||||
2026-03-14 11:46:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 59%| 322/546 [Elapsed: 38:03 < Remaining: 25:39, 6.87s/it]
|
||||
2026-03-14 11:46:45 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=200, inflight=2)
|
||||
2026-03-14 11:47:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 60%| 330/546 [Elapsed: 39:03 < Remaining: 23:56, 6.65s/it]
|
||||
2026-03-14 11:47:45 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=192, inflight=2)
|
||||
2026-03-14 11:48:30 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=184, inflight=2)
|
||||
2026-03-14 11:48:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 63%| 346/546 [Elapsed: 40:03 < Remaining: 21:44, 6.52s/it]
|
||||
2026-03-14 11:49:24 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=176, inflight=2)
|
||||
2026-03-14 11:49:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 65%| 354/546 [Elapsed: 41:03 < Remaining: 21:05, 6.59s/it]
|
||||
2026-03-14 11:50:21 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=168, inflight=2)
|
||||
2026-03-14 11:50:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 66%| 362/546 [Elapsed: 42:03 < Remaining: 20:42, 6.75s/it]
|
||||
2026-03-14 11:51:07 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=160, inflight=2)
|
||||
2026-03-14 11:51:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 68%| 370/546 [Elapsed: 43:03 < Remaining: 18:56, 6.45s/it]
|
||||
2026-03-14 11:52:09 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=152, inflight=2)
|
||||
2026-03-14 11:52:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 69%| 378/546 [Elapsed: 44:03 < Remaining: 19:09, 6.85s/it]
|
||||
2026-03-14 11:52:48 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=144, inflight=2)
|
||||
2026-03-14 11:53:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 71%| 386/546 [Elapsed: 45:03 < Remaining: 16:40, 6.26s/it]
|
||||
2026-03-14 11:53:52 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=136, inflight=2)
|
||||
2026-03-14 11:54:37 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=128, inflight=2)
|
||||
2026-03-14 11:54:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 74%| 402/546 [Elapsed: 46:03 < Remaining: 15:28, 6.45s/it]
|
||||
2026-03-14 11:55:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 74%| 402/546 [Elapsed: 47:03 < Remaining: 15:28, 6.45s/it]
|
||||
2026-03-14 11:55:43 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=120, inflight=2)
|
||||
2026-03-14 11:56:29 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=112, inflight=2)
|
||||
2026-03-14 11:56:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 77%| 418/546 [Elapsed: 48:03 < Remaining: 14:08, 6.63s/it]
|
||||
2026-03-14 11:57:26 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=104, inflight=2)
|
||||
2026-03-14 11:57:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 78%| 426/546 [Elapsed: 49:03 < Remaining: 13:29, 6.74s/it]
|
||||
2026-03-14 11:58:25 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=96, inflight=2)
|
||||
2026-03-14 11:58:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 79%| 434/546 [Elapsed: 50:03 < Remaining: 12:56, 6.93s/it]
|
||||
2026-03-14 11:59:11 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=88, inflight=2)
|
||||
2026-03-14 11:59:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 81%| 442/546 [Elapsed: 51:03 < Remaining: 11:24, 6.58s/it]
|
||||
2026-03-14 12:00:18 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=80, inflight=2)
|
||||
2026-03-14 12:00:41 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 82%| 450/546 [Elapsed: 52:03 < Remaining: 11:23, 7.12s/it]
|
||||
2026-03-14 12:01:00 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=72, inflight=2)
|
||||
2026-03-14 12:01:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 84%| 458/546 [Elapsed: 53:03 < Remaining: 09:37, 6.56s/it]
|
||||
2026-03-14 12:02:07 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=64, inflight=2)
|
||||
2026-03-14 12:02:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 85%| 466/546 [Elapsed: 54:03 < Remaining: 09:28, 7.11s/it]
|
||||
2026-03-14 12:02:55 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=56, inflight=2)
|
||||
2026-03-14 12:03:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 87%| 474/546 [Elapsed: 55:03 < Remaining: 08:07, 6.78s/it]
|
||||
2026-03-14 12:03:59 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=48, inflight=2)
|
||||
2026-03-14 12:04:36 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=40, inflight=2)
|
||||
2026-03-14 12:04:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 90%| 490/546 [Elapsed: 56:03 < Remaining: 05:55, 6.35s/it]
|
||||
2026-03-14 12:05:33 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=32, inflight=2)
|
||||
2026-03-14 12:05:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 91%| 498/546 [Elapsed: 57:03 < Remaining: 05:17, 6.62s/it]
|
||||
2026-03-14 12:06:23 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=24, inflight=2)
|
||||
2026-03-14 12:06:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 93%| 506/546 [Elapsed: 58:03 < Remaining: 04:20, 6.51s/it]
|
||||
2026-03-14 12:07:17 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=16, inflight=2)
|
||||
2026-03-14 12:07:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 94%| 514/546 [Elapsed: 59:03 < Remaining: 03:29, 6.55s/it]
|
||||
2026-03-14 12:08:08 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=8, inflight=2)
|
||||
2026-03-14 12:08:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 96%| 522/546 [Elapsed: 1:00:03 < Remaining: 02:35, 6.50s/it]
|
||||
2026-03-14 12:09:06 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=0, inflight=2)
|
||||
2026-03-14 12:09:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 97%| 530/546 [Elapsed: 1:01:04 < Remaining: 01:47, 6.72s/it]
|
||||
2026-03-14 12:10:42 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 99%| 538/546 [Elapsed: 1:02:04 < Remaining: 00:52, 6.51s/it]
|
||||
2026-03-14 12:10:51 - evalscope - INFO: Predicting[gpqa_extend@gpqa]: 100%| 546/546 [Elapsed: 1:02:12 < Remaining: 00:00, 6.69s/it]
|
||||
2026-03-14 12:10:51 - evalscope - INFO: Finished getting predictions for subset: gpqa.
|
||||
2026-03-14 12:10:51 - evalscope - INFO: Getting reviews for subset: gpqa
|
||||
2026-03-14 12:10:51 - evalscope - INFO: Reviewing 546 samples, if data is large, it may take a while.
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Reviewing[gpqa_extend@gpqa]: 100%| 546/546 [Elapsed: 00:00 < Remaining: 00:00, 822.90it/s]
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Finished reviewing subset: gpqa. Total reviewed: 546
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Aggregating scores for subset: gpqa
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Evaluating [gpqa_extend] 100%| 1/1 [Elapsed: 1:02:14 < Remaining: 00:00, 3734.34s/subset]
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Generating report...
|
||||
2026-03-14 12:10:52 - evalscope - INFO:
|
||||
gpqa_extend report table:
|
||||
+-----------------------------------------+-------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+=============+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | gpqa_extend | mean_acc | gpqa | 546 | 0.2454 | default |
|
||||
+-----------------------------------------+-------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110837/reports/llama3_3b_instruct_vallina_full_sft_30k/gpqa_extend.json
|
||||
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Benchmark gpqa_extend evaluation finished.
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+=============+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | gpqa_extend | mean_acc | gpqa | 546 | 0.2454 | default |
|
||||
+-----------------------------------------+-------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['gpqa_extend']
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA/20260314_110837
|
||||
2026-03-14 12:10:52 - evalscope - INFO: [进度条] GPQA 评测完成 ✓
|
||||
2026-03-14 12:10:52 - evalscope - INFO: [断点续传] GPQA 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GPQA_result.json
|
||||
2026-03-14 12:10:52 - evalscope - INFO: 完成评测 GPQA (7/8)
|
||||
2026-03-14 12:10:52 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-14 12:10:52 - evalscope - INFO: 正在评估 MMLUProNoMath (repeat: 1次) (剩余: 0个)
|
||||
2026-03-14 12:10:52 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-14 12:10:52 - evalscope - INFO: 开始创建 benchmark MMLUProNoMath 的 TaskConfig
|
||||
2026-03-14 12:10:52 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 12:10:52 - evalscope - INFO: [MMLUProNoMath] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['mmlu_pro_no_math'], eval_batch_size=2048
|
||||
2026-03-14 12:10:52 - evalscope - INFO: 开始评测 MMLUProNoMath...
|
||||
2026-03-14 12:10:52 - evalscope - INFO: [进度条] MMLUProNoMath 开始评测
|
||||
2026-03-14 12:10:52 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3d8b7e045c162f96f1d85b20b5c5915da1ca60da472d0312e903a2faa28ded5c
|
||||
size 24033160
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gpqa_extend",
|
||||
"dataset_name": "gpqa_extend",
|
||||
"dataset_pretty_name": "GPQA-Extended",
|
||||
"dataset_description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.2454,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "gpqa",
|
||||
"score": 0.2454,
|
||||
"num": 546
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0fa23efa774138e1b20f71d90440a8f958b14c32e1ed4bae4a094e4c7fc511de
|
||||
size 12572227
|
||||
36
evalscope/version_20260313_110435/run_1/GPQA_result.json
Normal file
36
evalscope/version_20260313_110435/run_1/GPQA_result.json
Normal file
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"gpqa_extend": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gpqa_extend",
|
||||
"dataset_name": "gpqa_extend",
|
||||
"dataset_pretty_name": "GPQA-Extended",
|
||||
"dataset_description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.2454,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "gpqa",
|
||||
"score": 0.2454,
|
||||
"num": 546
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
gsm8k:
|
||||
aggregation: mean
|
||||
dataset_id: AI-ModelScope/gsm8k
|
||||
default_subset: default
|
||||
description: GSM8K (Grade School Math 8K) is a dataset of grade school math problems,
|
||||
designed to evaluate the mathematical reasoning abilities of AI models.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: 'Here are some examples of how to solve similar problems:
|
||||
|
||||
|
||||
{fewshot}
|
||||
|
||||
|
||||
{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: gsm8k
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: GSM8K
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- main
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: train
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- gsm8k
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511
|
||||
@@ -0,0 +1,394 @@
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511/configs/task_config_79a30d.yaml
|
||||
2026-03-13 22:55:11 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"gsm8k"
|
||||
],
|
||||
"dataset_args": {
|
||||
"gsm8k": {
|
||||
"name": "gsm8k",
|
||||
"dataset_id": "AI-ModelScope/gsm8k",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"main"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": "train",
|
||||
"eval_split": "test",
|
||||
"prompt_template": "{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"few_shot_prompt_template": "Here are some examples of how to solve similar problems:\n\n{fewshot}\n\n{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "GSM8K",
|
||||
"description": "GSM8K (Grade School Math 8K) is a dataset of grade school math problems, designed to evaluate the mathematical reasoning abilities of AI models.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Start loading benchmark dataset: gsm8k
|
||||
2026-03-13 22:55:11 - evalscope - INFO: Loading dataset AI-ModelScope/gsm8k from modelscope > subset: main > split: test ...
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Start evaluating 1 subsets of the gsm8k: ['main']
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Evaluating subset: main
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Getting predictions for subset: main
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Processing 1319 samples, if data is large, it may take a while.
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 22:55:20 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 22:55:29 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1309, inflight=2)
|
||||
2026-03-13 22:55:31 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1301, inflight=2)
|
||||
2026-03-13 22:56:21 - evalscope - INFO: Predicting[gsm8k@main]: 0%| 2/1319 [Elapsed: 01:00 < Remaining: 1:51:27, 5.08s/it]
|
||||
2026-03-13 22:56:45 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1293, inflight=2)
|
||||
2026-03-13 22:57:03 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1285, inflight=2)
|
||||
2026-03-13 22:57:21 - evalscope - INFO: Predicting[gsm8k@main]: 1%| 18/1319 [Elapsed: 02:00 < Remaining: 2:49:37, 7.82s/it]
|
||||
2026-03-13 22:57:44 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1277, inflight=2)
|
||||
2026-03-13 22:58:04 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1269, inflight=2)
|
||||
2026-03-13 22:58:22 - evalscope - INFO: Predicting[gsm8k@main]: 3%| 34/1319 [Elapsed: 03:00 < Remaining: 1:39:18, 4.64s/it]
|
||||
2026-03-13 22:59:06 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1261, inflight=2)
|
||||
2026-03-13 22:59:20 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1253, inflight=2)
|
||||
2026-03-13 22:59:22 - evalscope - INFO: Predicting[gsm8k@main]: 4%| 50/1319 [Elapsed: 04:00 < Remaining: 1:33:13, 4.41s/it]
|
||||
2026-03-13 22:59:50 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1245, inflight=2)
|
||||
2026-03-13 23:00:22 - evalscope - INFO: Predicting[gsm8k@main]: 4%| 58/1319 [Elapsed: 05:01 < Remaining: 1:28:05, 4.19s/it]
|
||||
2026-03-13 23:01:00 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1237, inflight=2)
|
||||
2026-03-13 23:01:11 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1229, inflight=2)
|
||||
2026-03-13 23:01:22 - evalscope - INFO: Predicting[gsm8k@main]: 6%| 74/1319 [Elapsed: 06:01 < Remaining: 1:29:51, 4.33s/it]
|
||||
2026-03-13 23:01:58 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1221, inflight=2)
|
||||
2026-03-13 23:02:22 - evalscope - INFO: Predicting[gsm8k@main]: 6%| 82/1319 [Elapsed: 07:01 < Remaining: 1:38:30, 4.78s/it]
|
||||
2026-03-13 23:02:25 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1213, inflight=2)
|
||||
2026-03-13 23:02:56 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1205, inflight=2)
|
||||
2026-03-13 23:03:23 - evalscope - INFO: Predicting[gsm8k@main]: 7%| 98/1319 [Elapsed: 08:01 < Remaining: 1:25:38, 4.21s/it]
|
||||
2026-03-13 23:03:35 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1197, inflight=2)
|
||||
2026-03-13 23:04:14 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1189, inflight=2)
|
||||
2026-03-13 23:04:23 - evalscope - INFO: Predicting[gsm8k@main]: 9%| 114/1319 [Elapsed: 09:01 < Remaining: 1:30:49, 4.52s/it]
|
||||
2026-03-13 23:04:32 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1181, inflight=2)
|
||||
2026-03-13 23:05:14 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1173, inflight=2)
|
||||
2026-03-13 23:05:23 - evalscope - INFO: Predicting[gsm8k@main]: 10%| 130/1319 [Elapsed: 10:02 < Remaining: 1:25:20, 4.31s/it]
|
||||
2026-03-13 23:05:31 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1165, inflight=2)
|
||||
2026-03-13 23:05:44 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1157, inflight=2)
|
||||
2026-03-13 23:06:23 - evalscope - INFO: Predicting[gsm8k@main]: 11%| 146/1319 [Elapsed: 11:02 < Remaining: 59:44, 3.06s/it]
|
||||
2026-03-13 23:06:43 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1149, inflight=2)
|
||||
2026-03-13 23:06:58 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1141, inflight=2)
|
||||
2026-03-13 23:07:21 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1133, inflight=2)
|
||||
2026-03-13 23:07:23 - evalscope - INFO: Predicting[gsm8k@main]: 13%| 170/1319 [Elapsed: 12:02 < Remaining: 1:04:41, 3.38s/it]
|
||||
2026-03-13 23:08:06 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1125, inflight=2)
|
||||
2026-03-13 23:08:24 - evalscope - INFO: Predicting[gsm8k@main]: 13%| 178/1319 [Elapsed: 13:02 < Remaining: 1:17:09, 4.06s/it]
|
||||
2026-03-13 23:08:28 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1117, inflight=2)
|
||||
2026-03-13 23:08:53 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1109, inflight=2)
|
||||
2026-03-13 23:09:24 - evalscope - INFO: Predicting[gsm8k@main]: 15%| 194/1319 [Elapsed: 14:02 < Remaining: 1:05:46, 3.51s/it]
|
||||
2026-03-13 23:09:26 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1101, inflight=2)
|
||||
2026-03-13 23:10:02 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1093, inflight=2)
|
||||
2026-03-13 23:10:24 - evalscope - INFO: Predicting[gsm8k@main]: 16%| 210/1319 [Elapsed: 15:03 < Remaining: 1:12:51, 3.94s/it]
|
||||
2026-03-13 23:10:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1085, inflight=2)
|
||||
2026-03-13 23:11:07 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1077, inflight=2)
|
||||
2026-03-13 23:11:24 - evalscope - INFO: Predicting[gsm8k@main]: 17%| 226/1319 [Elapsed: 16:03 < Remaining: 1:09:47, 3.83s/it]
|
||||
2026-03-13 23:11:27 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1069, inflight=2)
|
||||
2026-03-13 23:12:21 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1061, inflight=2)
|
||||
2026-03-13 23:12:24 - evalscope - INFO: Predicting[gsm8k@main]: 18%| 242/1319 [Elapsed: 17:03 < Remaining: 1:19:36, 4.43s/it]
|
||||
2026-03-13 23:12:46 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1053, inflight=2)
|
||||
2026-03-13 23:13:24 - evalscope - INFO: Predicting[gsm8k@main]: 19%| 250/1319 [Elapsed: 18:03 < Remaining: 1:12:43, 4.08s/it]
|
||||
2026-03-13 23:13:34 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1045, inflight=2)
|
||||
2026-03-13 23:13:59 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1037, inflight=2)
|
||||
2026-03-13 23:14:06 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1029, inflight=2)
|
||||
2026-03-13 23:14:25 - evalscope - INFO: Predicting[gsm8k@main]: 21%| 274/1319 [Elapsed: 19:03 < Remaining: 54:53, 3.15s/it]
|
||||
2026-03-13 23:14:36 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1021, inflight=2)
|
||||
2026-03-13 23:14:58 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=1013, inflight=2)
|
||||
2026-03-13 23:15:25 - evalscope - INFO: Predicting[gsm8k@main]: 22%| 290/1319 [Elapsed: 20:03 < Remaining: 54:40, 3.19s/it]
|
||||
2026-03-13 23:15:25 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=1005, inflight=2)
|
||||
2026-03-13 23:16:07 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=997, inflight=2)
|
||||
2026-03-13 23:16:22 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=989, inflight=2)
|
||||
2026-03-13 23:16:25 - evalscope - INFO: Predicting[gsm8k@main]: 24%| 314/1319 [Elapsed: 21:04 < Remaining: 53:59, 3.22s/it]
|
||||
2026-03-13 23:16:59 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=981, inflight=2)
|
||||
2026-03-13 23:17:25 - evalscope - INFO: Predicting[gsm8k@main]: 24%| 322/1319 [Elapsed: 22:04 < Remaining: 1:00:36, 3.65s/it]
|
||||
2026-03-13 23:17:43 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=973, inflight=2)
|
||||
2026-03-13 23:17:47 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=965, inflight=2)
|
||||
2026-03-13 23:18:25 - evalscope - INFO: Predicting[gsm8k@main]: 26%| 338/1319 [Elapsed: 23:04 < Remaining: 50:37, 3.10s/it]
|
||||
2026-03-13 23:18:40 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=957, inflight=2)
|
||||
2026-03-13 23:18:53 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=949, inflight=2)
|
||||
2026-03-13 23:19:25 - evalscope - INFO: Predicting[gsm8k@main]: 27%| 354/1319 [Elapsed: 24:04 < Remaining: 54:41, 3.40s/it]
|
||||
2026-03-13 23:19:50 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=941, inflight=2)
|
||||
2026-03-13 23:20:17 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=933, inflight=2)
|
||||
2026-03-13 23:20:26 - evalscope - INFO: Predicting[gsm8k@main]: 28%| 370/1319 [Elapsed: 25:04 < Remaining: 1:06:08, 4.18s/it]
|
||||
2026-03-13 23:21:26 - evalscope - INFO: Predicting[gsm8k@main]: 28%| 370/1319 [Elapsed: 26:04 < Remaining: 1:06:08, 4.18s/it]
|
||||
2026-03-13 23:21:28 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=925, inflight=2)
|
||||
2026-03-13 23:21:33 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=917, inflight=2)
|
||||
2026-03-13 23:22:26 - evalscope - INFO: Predicting[gsm8k@main]: 29%| 386/1319 [Elapsed: 27:05 < Remaining: 1:04:01, 4.12s/it]
|
||||
2026-03-13 23:22:26 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=909, inflight=2)
|
||||
2026-03-13 23:23:00 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=901, inflight=2)
|
||||
2026-03-13 23:23:08 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=893, inflight=2)
|
||||
2026-03-13 23:23:26 - evalscope - INFO: Predicting[gsm8k@main]: 31%| 410/1319 [Elapsed: 28:05 < Remaining: 53:54, 3.56s/it]
|
||||
2026-03-13 23:23:48 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=885, inflight=2)
|
||||
2026-03-13 23:24:12 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=877, inflight=2)
|
||||
2026-03-13 23:24:26 - evalscope - INFO: Predicting[gsm8k@main]: 32%| 426/1319 [Elapsed: 29:05 < Remaining: 54:52, 3.69s/it]
|
||||
2026-03-13 23:24:48 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=869, inflight=2)
|
||||
2026-03-13 23:25:26 - evalscope - INFO: Predicting[gsm8k@main]: 33%| 434/1319 [Elapsed: 30:05 < Remaining: 58:01, 3.93s/it]
|
||||
2026-03-13 23:25:50 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=861, inflight=2)
|
||||
2026-03-13 23:25:50 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=853, inflight=2)
|
||||
2026-03-13 23:26:26 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=845, inflight=2)
|
||||
2026-03-13 23:26:27 - evalscope - INFO: Predicting[gsm8k@main]: 34%| 451/1319 [Elapsed: 31:05 < Remaining: 55:27, 3.83s/it]
|
||||
2026-03-13 23:27:04 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=837, inflight=2)
|
||||
2026-03-13 23:27:27 - evalscope - INFO: Predicting[gsm8k@main]: 35%| 466/1319 [Elapsed: 32:05 < Remaining: 58:27, 4.11s/it]
|
||||
2026-03-13 23:27:37 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=829, inflight=2)
|
||||
2026-03-13 23:27:54 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=821, inflight=2)
|
||||
2026-03-13 23:27:57 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=813, inflight=2)
|
||||
2026-03-13 23:28:27 - evalscope - INFO: Predicting[gsm8k@main]: 37%| 490/1319 [Elapsed: 33:05 < Remaining: 35:06, 2.54s/it]
|
||||
2026-03-13 23:28:45 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=805, inflight=2)
|
||||
2026-03-13 23:29:05 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=797, inflight=2)
|
||||
2026-03-13 23:29:27 - evalscope - INFO: Predicting[gsm8k@main]: 38%| 506/1319 [Elapsed: 34:06 < Remaining: 44:10, 3.26s/it]
|
||||
2026-03-13 23:30:16 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=789, inflight=2)
|
||||
2026-03-13 23:30:27 - evalscope - INFO: Predicting[gsm8k@main]: 39%| 514/1319 [Elapsed: 35:06 < Remaining: 1:06:55, 4.99s/it]
|
||||
2026-03-13 23:30:37 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=781, inflight=2)
|
||||
2026-03-13 23:31:27 - evalscope - INFO: Predicting[gsm8k@main]: 40%| 522/1319 [Elapsed: 36:06 < Remaining: 56:51, 4.28s/it]
|
||||
2026-03-13 23:31:50 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=773, inflight=2)
|
||||
2026-03-13 23:32:12 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=765, inflight=2)
|
||||
2026-03-13 23:32:27 - evalscope - INFO: Predicting[gsm8k@main]: 41%| 538/1319 [Elapsed: 37:06 < Remaining: 1:02:34, 4.81s/it]
|
||||
2026-03-13 23:32:50 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=757, inflight=2)
|
||||
2026-03-13 23:32:51 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=749, inflight=2)
|
||||
2026-03-13 23:33:28 - evalscope - INFO: Predicting[gsm8k@main]: 42%| 554/1319 [Elapsed: 38:06 < Remaining: 43:15, 3.39s/it]
|
||||
2026-03-13 23:33:50 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=741, inflight=2)
|
||||
2026-03-13 23:33:52 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=733, inflight=2)
|
||||
2026-03-13 23:34:28 - evalscope - INFO: Predicting[gsm8k@main]: 43%| 570/1319 [Elapsed: 39:06 < Remaining: 41:04, 3.29s/it]
|
||||
2026-03-13 23:34:37 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=725, inflight=2)
|
||||
2026-03-13 23:34:57 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=717, inflight=2)
|
||||
2026-03-13 23:35:28 - evalscope - INFO: Predicting[gsm8k@main]: 44%| 586/1319 [Elapsed: 40:06 < Remaining: 43:20, 3.55s/it]
|
||||
2026-03-13 23:36:13 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=709, inflight=2)
|
||||
2026-03-13 23:36:20 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=701, inflight=2)
|
||||
2026-03-13 23:36:28 - evalscope - INFO: Predicting[gsm8k@main]: 46%| 602/1319 [Elapsed: 41:07 < Remaining: 47:29, 3.97s/it]
|
||||
2026-03-13 23:36:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=693, inflight=2)
|
||||
2026-03-13 23:37:28 - evalscope - INFO: Predicting[gsm8k@main]: 46%| 610/1319 [Elapsed: 42:07 < Remaining: 45:45, 3.87s/it]
|
||||
2026-03-13 23:37:44 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=685, inflight=2)
|
||||
2026-03-13 23:38:00 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=677, inflight=2)
|
||||
2026-03-13 23:38:28 - evalscope - INFO: Predicting[gsm8k@main]: 47%| 626/1319 [Elapsed: 43:07 < Remaining: 45:33, 3.95s/it]
|
||||
2026-03-13 23:38:40 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=669, inflight=2)
|
||||
2026-03-13 23:39:10 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=661, inflight=2)
|
||||
2026-03-13 23:39:19 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=653, inflight=2)
|
||||
2026-03-13 23:39:28 - evalscope - INFO: Predicting[gsm8k@main]: 49%| 650/1319 [Elapsed: 44:07 < Remaining: 36:04, 3.24s/it]
|
||||
2026-03-13 23:39:53 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=645, inflight=2)
|
||||
2026-03-13 23:40:19 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=637, inflight=2)
|
||||
2026-03-13 23:40:28 - evalscope - INFO: Predicting[gsm8k@main]: 50%| 666/1319 [Elapsed: 45:07 < Remaining: 37:20, 3.43s/it]
|
||||
2026-03-13 23:41:05 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=629, inflight=2)
|
||||
2026-03-13 23:41:25 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=621, inflight=2)
|
||||
2026-03-13 23:41:29 - evalscope - INFO: Predicting[gsm8k@main]: 52%| 682/1319 [Elapsed: 46:07 < Remaining: 38:40, 3.64s/it]
|
||||
2026-03-13 23:42:05 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=613, inflight=2)
|
||||
2026-03-13 23:42:29 - evalscope - INFO: Predicting[gsm8k@main]: 52%| 690/1319 [Elapsed: 47:07 < Remaining: 42:29, 4.05s/it]
|
||||
2026-03-13 23:42:54 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=605, inflight=2)
|
||||
2026-03-13 23:43:13 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=597, inflight=2)
|
||||
2026-03-13 23:43:29 - evalscope - INFO: Predicting[gsm8k@main]: 54%| 706/1319 [Elapsed: 48:07 < Remaining: 40:21, 3.95s/it]
|
||||
2026-03-13 23:43:38 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=589, inflight=2)
|
||||
2026-03-13 23:44:01 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=581, inflight=2)
|
||||
2026-03-13 23:44:28 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=573, inflight=2)
|
||||
2026-03-13 23:44:29 - evalscope - INFO: Predicting[gsm8k@main]: 55%| 723/1319 [Elapsed: 49:07 < Remaining: 34:18, 3.45s/it]
|
||||
2026-03-13 23:44:52 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=565, inflight=2)
|
||||
2026-03-13 23:45:29 - evalscope - INFO: Predicting[gsm8k@main]: 56%| 738/1319 [Elapsed: 50:08 < Remaining: 32:08, 3.32s/it]
|
||||
2026-03-13 23:45:41 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=557, inflight=2)
|
||||
2026-03-13 23:46:01 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=549, inflight=2)
|
||||
2026-03-13 23:46:29 - evalscope - INFO: Predicting[gsm8k@main]: 57%| 754/1319 [Elapsed: 51:08 < Remaining: 34:16, 3.64s/it]
|
||||
2026-03-13 23:46:37 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=541, inflight=2)
|
||||
2026-03-13 23:46:46 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=533, inflight=2)
|
||||
2026-03-13 23:47:29 - evalscope - INFO: Predicting[gsm8k@main]: 58%| 770/1319 [Elapsed: 52:08 < Remaining: 28:04, 3.07s/it]
|
||||
2026-03-13 23:47:53 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=525, inflight=2)
|
||||
2026-03-13 23:47:53 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=517, inflight=2)
|
||||
2026-03-13 23:48:29 - evalscope - INFO: Predicting[gsm8k@main]: 60%| 786/1319 [Elapsed: 53:08 < Remaining: 41:26, 4.66s/it]
|
||||
2026-03-13 23:48:57 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=509, inflight=2)
|
||||
2026-03-13 23:49:26 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=501, inflight=2)
|
||||
2026-03-13 23:49:29 - evalscope - INFO: Predicting[gsm8k@main]: 61%| 802/1319 [Elapsed: 54:08 < Remaining: 36:01, 4.18s/it]
|
||||
2026-03-13 23:49:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=493, inflight=2)
|
||||
2026-03-13 23:50:11 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=485, inflight=2)
|
||||
2026-03-13 23:50:29 - evalscope - INFO: Predicting[gsm8k@main]: 62%| 818/1319 [Elapsed: 55:08 < Remaining: 29:41, 3.56s/it]
|
||||
2026-03-13 23:50:35 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=477, inflight=2)
|
||||
2026-03-13 23:51:24 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=469, inflight=2)
|
||||
2026-03-13 23:51:30 - evalscope - INFO: Predicting[gsm8k@main]: 63%| 834/1319 [Elapsed: 56:08 < Remaining: 33:48, 4.18s/it]
|
||||
2026-03-13 23:51:39 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=461, inflight=2)
|
||||
2026-03-13 23:52:27 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=453, inflight=2)
|
||||
2026-03-13 23:52:30 - evalscope - INFO: Predicting[gsm8k@main]: 64%| 850/1319 [Elapsed: 57:08 < Remaining: 33:10, 4.24s/it]
|
||||
2026-03-13 23:53:00 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=445, inflight=2)
|
||||
2026-03-13 23:53:30 - evalscope - INFO: Predicting[gsm8k@main]: 65%| 858/1319 [Elapsed: 58:08 < Remaining: 32:21, 4.21s/it]
|
||||
2026-03-13 23:53:32 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=437, inflight=2)
|
||||
2026-03-13 23:53:52 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=429, inflight=2)
|
||||
2026-03-13 23:54:29 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=421, inflight=2)
|
||||
2026-03-13 23:54:30 - evalscope - INFO: Predicting[gsm8k@main]: 67%| 882/1319 [Elapsed: 59:08 < Remaining: 28:29, 3.91s/it]
|
||||
2026-03-13 23:54:57 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=413, inflight=2)
|
||||
2026-03-13 23:55:30 - evalscope - INFO: Predicting[gsm8k@main]: 67%| 890/1319 [Elapsed: 1:00:09 < Remaining: 27:22, 3.83s/it]
|
||||
2026-03-13 23:55:41 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=405, inflight=2)
|
||||
2026-03-13 23:55:53 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=397, inflight=2)
|
||||
2026-03-13 23:56:14 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=389, inflight=2)
|
||||
2026-03-13 23:56:25 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=381, inflight=2)
|
||||
2026-03-13 23:56:30 - evalscope - INFO: Predicting[gsm8k@main]: 70%| 922/1319 [Elapsed: 1:01:09 < Remaining: 17:46, 2.69s/it]
|
||||
2026-03-13 23:57:21 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=373, inflight=2)
|
||||
2026-03-13 23:57:30 - evalscope - INFO: Predicting[gsm8k@main]: 71%| 930/1319 [Elapsed: 1:02:09 < Remaining: 25:49, 3.98s/it]
|
||||
2026-03-13 23:57:45 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=365, inflight=2)
|
||||
2026-03-13 23:58:30 - evalscope - INFO: Predicting[gsm8k@main]: 71%| 938/1319 [Elapsed: 1:03:09 < Remaining: 23:11, 3.65s/it]
|
||||
2026-03-13 23:58:40 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=357, inflight=2)
|
||||
2026-03-13 23:59:20 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=349, inflight=2)
|
||||
2026-03-13 23:59:30 - evalscope - INFO: Predicting[gsm8k@main]: 72%| 954/1319 [Elapsed: 1:04:09 < Remaining: 28:48, 4.74s/it]
|
||||
2026-03-14 00:00:10 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=341, inflight=2)
|
||||
2026-03-14 00:00:12 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=333, inflight=2)
|
||||
2026-03-14 00:00:30 - evalscope - INFO: Predicting[gsm8k@main]: 74%| 970/1319 [Elapsed: 1:05:09 < Remaining: 21:35, 3.71s/it]
|
||||
2026-03-14 00:00:56 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=325, inflight=2)
|
||||
2026-03-14 00:01:30 - evalscope - INFO: Predicting[gsm8k@main]: 74%| 978/1319 [Elapsed: 1:06:09 < Remaining: 24:09, 4.25s/it]
|
||||
2026-03-14 00:01:43 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=317, inflight=2)
|
||||
2026-03-14 00:01:52 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=309, inflight=2)
|
||||
2026-03-14 00:02:31 - evalscope - INFO: Predicting[gsm8k@main]: 75%| 994/1319 [Elapsed: 1:07:09 < Remaining: 19:48, 3.66s/it]
|
||||
2026-03-14 00:02:47 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=301, inflight=2)
|
||||
2026-03-14 00:02:58 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=293, inflight=2)
|
||||
2026-03-14 00:03:31 - evalscope - INFO: Predicting[gsm8k@main]: 77%| 1010/1319 [Elapsed: 1:08:09 < Remaining: 18:47, 3.65s/it]
|
||||
2026-03-14 00:03:53 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=285, inflight=2)
|
||||
2026-03-14 00:04:30 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=277, inflight=2)
|
||||
2026-03-14 00:04:31 - evalscope - INFO: Predicting[gsm8k@main]: 77%| 1019/1319 [Elapsed: 1:09:09 < Remaining: 23:06, 4.62s/it]
|
||||
2026-03-14 00:04:36 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=269, inflight=2)
|
||||
2026-03-14 00:05:31 - evalscope - INFO: Predicting[gsm8k@main]: 78%| 1034/1319 [Elapsed: 1:10:09 < Remaining: 16:26, 3.46s/it]
|
||||
2026-03-14 00:06:05 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=261, inflight=2)
|
||||
2026-03-14 00:06:11 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=253, inflight=2)
|
||||
2026-03-14 00:06:31 - evalscope - INFO: Predicting[gsm8k@main]: 80%| 1050/1319 [Elapsed: 1:11:09 < Remaining: 19:06, 4.26s/it]
|
||||
2026-03-14 00:06:51 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=245, inflight=2)
|
||||
2026-03-14 00:07:31 - evalscope - INFO: Predicting[gsm8k@main]: 80%| 1058/1319 [Elapsed: 1:12:10 < Remaining: 19:30, 4.48s/it]
|
||||
2026-03-14 00:07:35 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=237, inflight=2)
|
||||
2026-03-14 00:08:08 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=229, inflight=2)
|
||||
2026-03-14 00:08:31 - evalscope - INFO: Predicting[gsm8k@main]: 81%| 1074/1319 [Elapsed: 1:13:10 < Remaining: 18:35, 4.55s/it]
|
||||
2026-03-14 00:08:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=221, inflight=2)
|
||||
2026-03-14 00:09:01 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=213, inflight=2)
|
||||
2026-03-14 00:09:31 - evalscope - INFO: Predicting[gsm8k@main]: 83%| 1090/1319 [Elapsed: 1:14:10 < Remaining: 14:29, 3.80s/it]
|
||||
2026-03-14 00:09:36 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=205, inflight=2)
|
||||
2026-03-14 00:09:43 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=197, inflight=2)
|
||||
2026-03-14 00:10:25 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=189, inflight=2)
|
||||
2026-03-14 00:10:31 - evalscope - INFO: Predicting[gsm8k@main]: 84%| 1114/1319 [Elapsed: 1:15:10 < Remaining: 12:41, 3.71s/it]
|
||||
2026-03-14 00:10:38 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=181, inflight=2)
|
||||
2026-03-14 00:11:31 - evalscope - INFO: Predicting[gsm8k@main]: 85%| 1122/1319 [Elapsed: 1:16:10 < Remaining: 10:01, 3.05s/it]
|
||||
2026-03-14 00:12:00 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=173, inflight=2)
|
||||
2026-03-14 00:12:13 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=165, inflight=2)
|
||||
2026-03-14 00:12:31 - evalscope - INFO: Predicting[gsm8k@main]: 86%| 1138/1319 [Elapsed: 1:17:10 < Remaining: 12:28, 4.14s/it]
|
||||
2026-03-14 00:12:46 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=157, inflight=2)
|
||||
2026-03-14 00:13:31 - evalscope - INFO: Predicting[gsm8k@main]: 87%| 1146/1319 [Elapsed: 1:18:10 < Remaining: 11:55, 4.13s/it]
|
||||
2026-03-14 00:13:42 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=149, inflight=2)
|
||||
2026-03-14 00:13:46 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=141, inflight=2)
|
||||
2026-03-14 00:14:31 - evalscope - INFO: Predicting[gsm8k@main]: 88%| 1162/1319 [Elapsed: 1:19:10 < Remaining: 09:38, 3.69s/it]
|
||||
2026-03-14 00:14:59 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=133, inflight=2)
|
||||
2026-03-14 00:15:19 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=125, inflight=2)
|
||||
2026-03-14 00:15:31 - evalscope - INFO: Predicting[gsm8k@main]: 89%| 1178/1319 [Elapsed: 1:20:10 < Remaining: 10:25, 4.44s/it]
|
||||
2026-03-14 00:16:27 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=117, inflight=2)
|
||||
2026-03-14 00:16:31 - evalscope - INFO: Predicting[gsm8k@main]: 90%| 1186/1319 [Elapsed: 1:21:10 < Remaining: 12:32, 5.66s/it]
|
||||
2026-03-14 00:16:34 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=109, inflight=2)
|
||||
2026-03-14 00:17:31 - evalscope - INFO: Predicting[gsm8k@main]: 91%| 1194/1319 [Elapsed: 1:22:10 < Remaining: 08:47, 4.22s/it]
|
||||
2026-03-14 00:17:33 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=101, inflight=2)
|
||||
2026-03-14 00:17:48 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=93, inflight=2)
|
||||
2026-03-14 00:18:22 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=85, inflight=2)
|
||||
2026-03-14 00:18:32 - evalscope - INFO: Predicting[gsm8k@main]: 92%| 1218/1319 [Elapsed: 1:23:10 < Remaining: 07:04, 4.20s/it]
|
||||
2026-03-14 00:18:49 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=77, inflight=2)
|
||||
2026-03-14 00:18:59 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=69, inflight=2)
|
||||
2026-03-14 00:19:17 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=61, inflight=2)
|
||||
2026-03-14 00:19:32 - evalscope - INFO: Predicting[gsm8k@main]: 94%| 1242/1319 [Elapsed: 1:24:10 < Remaining: 03:41, 2.88s/it]
|
||||
2026-03-14 00:19:43 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=53, inflight=2)
|
||||
2026-03-14 00:20:08 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=45, inflight=2)
|
||||
2026-03-14 00:20:32 - evalscope - INFO: Predicting[gsm8k@main]: 95%| 1258/1319 [Elapsed: 1:25:10 < Remaining: 03:04, 3.03s/it]
|
||||
2026-03-14 00:21:26 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=37, inflight=2)
|
||||
2026-03-14 00:21:31 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=29, inflight=2)
|
||||
2026-03-14 00:21:32 - evalscope - INFO: Predicting[gsm8k@main]: 96%| 1267/1319 [Elapsed: 1:26:10 < Remaining: 03:13, 3.72s/it]
|
||||
2026-03-14 00:21:52 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=21, inflight=2)
|
||||
2026-03-14 00:22:21 - evalscope - INFO: Dispatcher: Worker-1 <- 8 prompts (pending=13, inflight=2)
|
||||
2026-03-14 00:22:32 - evalscope - INFO: Predicting[gsm8k@main]: 98%| 1290/1319 [Elapsed: 1:27:10 < Remaining: 01:40, 3.46s/it]
|
||||
2026-03-14 00:22:53 - evalscope - INFO: Dispatcher: Worker-0 <- 8 prompts (pending=5, inflight=2)
|
||||
2026-03-14 00:23:11 - evalscope - INFO: Dispatcher: Worker-1 <- 5 prompts (pending=0, inflight=2)
|
||||
2026-03-14 00:23:32 - evalscope - INFO: Predicting[gsm8k@main]: 99%| 1306/1319 [Elapsed: 1:28:10 < Remaining: 00:41, 3.21s/it]
|
||||
2026-03-14 00:24:24 - evalscope - INFO: Predicting[gsm8k@main]: 100%| 1319/1319 [Elapsed: 1:29:02 < Remaining: 00:00, 3.78s/it]
|
||||
2026-03-14 00:24:24 - evalscope - INFO: Finished getting predictions for subset: main.
|
||||
2026-03-14 00:24:24 - evalscope - INFO: Getting reviews for subset: main
|
||||
2026-03-14 00:24:24 - evalscope - INFO: Reviewing 1319 samples, if data is large, it may take a while.
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Reviewing[gsm8k@main]: 100%| 1319/1319 [Elapsed: 00:01 < Remaining: 00:00, 402.73it/s]
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Finished reviewing subset: main. Total reviewed: 1319
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Aggregating scores for subset: main
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Evaluating [gsm8k] 100%| 1/1 [Elapsed: 1:29:05 < Remaining: 00:00, 5345.55s/subset]
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Generating report...
|
||||
2026-03-14 00:24:26 - evalscope - INFO:
|
||||
gsm8k report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | gsm8k | mean_acc | main | 1319 | 0.7521 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511/reports/llama3_3b_instruct_vallina_full_sft_30k/gsm8k.json
|
||||
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Benchmark gsm8k evaluation finished.
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | gsm8k | mean_acc | main | 1319 | 0.7521 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['gsm8k']
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K/20260313_225511
|
||||
2026-03-14 00:24:26 - evalscope - INFO: [进度条] GSM8K 评测完成 ✓
|
||||
2026-03-14 00:24:26 - evalscope - INFO: [断点续传] GSM8K 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/GSM8K_result.json
|
||||
2026-03-14 00:24:26 - evalscope - INFO: 完成评测 GSM8K (6/8)
|
||||
2026-03-14 00:24:26 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-14 00:24:26 - evalscope - INFO: 正在评估 GPQA (repeat: 1次) (剩余: 1个)
|
||||
2026-03-14 00:24:26 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-14 00:24:26 - evalscope - INFO: 开始创建 benchmark GPQA 的 TaskConfig
|
||||
2026-03-14 00:24:26 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 00:24:26 - evalscope - INFO: [GPQA] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['gpqa_extend'], eval_batch_size=2048
|
||||
2026-03-14 00:24:26 - evalscope - INFO: 开始评测 GPQA...
|
||||
2026-03-14 00:24:26 - evalscope - INFO: [进度条] GPQA 开始评测
|
||||
2026-03-14 00:24:26 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2203fc04ce248639dbd9416a8180614326bcb8e09cd9d5176a83d4e42338a1ad
|
||||
size 30591469
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gsm8k",
|
||||
"dataset_name": "gsm8k",
|
||||
"dataset_pretty_name": "GSM8K",
|
||||
"dataset_description": "GSM8K (Grade School Math 8K) is a dataset of grade school math problems, designed to evaluate the mathematical reasoning abilities of AI models.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7521,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "main",
|
||||
"score": 0.7521,
|
||||
"num": 1319
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
36
evalscope/version_20260313_110435/run_1/GSM8K_result.json
Normal file
36
evalscope/version_20260313_110435/run_1/GSM8K_result.json
Normal file
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"gsm8k": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gsm8k",
|
||||
"dataset_name": "gsm8k",
|
||||
"dataset_pretty_name": "GSM8K",
|
||||
"dataset_description": "GSM8K (Grade School Math 8K) is a dataset of grade school math problems, designed to evaluate the mathematical reasoning abilities of AI models.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7521,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "main",
|
||||
"score": 0.7521,
|
||||
"num": 1319
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
math_500:
|
||||
aggregation: mean
|
||||
dataset_id: AI-ModelScope/MATH-500
|
||||
default_subset: default
|
||||
description: MATH-500 is a benchmark for evaluating mathematical reasoning capabilities
|
||||
of AI models. It consists of 500 diverse math problems across five levels of
|
||||
difficulty, designed to test a model's ability to solve complex mathematical
|
||||
problems by generating step-by-step solutions and providing the correct final
|
||||
answer.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: math_500
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: MATH-500
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- Level 1
|
||||
- Level 2
|
||||
- Level 3
|
||||
- Level 4
|
||||
- Level 5
|
||||
system_prompt: null
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- math_500
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500/20260313_142308
|
||||
@@ -0,0 +1,600 @@
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Running with native backend
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500/20260313_142308/configs/task_config_101b5e.yaml
|
||||
2026-03-13 14:23:08 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"math_500"
|
||||
],
|
||||
"dataset_args": {
|
||||
"math_500": {
|
||||
"name": "math_500",
|
||||
"dataset_id": "AI-ModelScope/MATH-500",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"Level 1",
|
||||
"Level 2",
|
||||
"Level 3",
|
||||
"Level 4",
|
||||
"Level 5"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": null,
|
||||
"eval_split": "test",
|
||||
"prompt_template": "{question}\nPlease reason step by step, and put your final answer within \\boxed{{}}.",
|
||||
"few_shot_prompt_template": null,
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "MATH-500",
|
||||
"description": "MATH-500 is a benchmark for evaluating mathematical reasoning capabilities of AI models. It consists of 500 diverse math problems across five levels of difficulty, designed to test a model's ability to solve complex mathematical problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"tags": [
|
||||
"Math",
|
||||
"Reasoning"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
{
|
||||
"acc": {
|
||||
"numeric": true
|
||||
}
|
||||
}
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": null,
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500/20260313_142308",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Start loading benchmark dataset: math_500
|
||||
2026-03-13 14:23:08 - evalscope - INFO: Loading dataset AI-ModelScope/MATH-500 from modelscope > subset: default > split: test ...
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Start evaluating 5 subsets of the math_500: ['Level 1', 'Level 2', 'Level 3', 'Level 4', 'Level 5']
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Evaluating subset: Level 1
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Getting predictions for subset: Level 1
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Processing 43 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 14:23:16 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 14:23:54 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 14:24:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 14:24:16 - evalscope - INFO: Predicting[math_500@Level 1]: 5%| 2/43 [Elapsed: 01:00 < Remaining: 14:43, 21.54s/it]
|
||||
2026-03-13 14:25:10 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 14:25:16 - evalscope - INFO: Predicting[math_500@Level 1]: 9%| 4/43 [Elapsed: 02:00 < Remaining: 27:12, 41.85s/it]
|
||||
2026-03-13 14:25:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 14:25:29 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 14:25:42 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 14:25:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 14:25:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 14:26:13 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 14:26:16 - evalscope - INFO: Predicting[math_500@Level 1]: 37%| 16/43 [Elapsed: 03:00 < Remaining: 03:14, 7.22s/it]
|
||||
2026-03-13 14:26:25 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 14:26:28 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 14:26:50 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 14:26:54 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 14:27:11 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 14:27:17 - evalscope - INFO: Predicting[math_500@Level 1]: 60%| 26/43 [Elapsed: 04:00 < Remaining: 01:48, 6.38s/it]
|
||||
2026-03-13 14:28:17 - evalscope - INFO: Predicting[math_500@Level 1]: 60%| 26/43 [Elapsed: 05:00 < Remaining: 01:48, 6.38s/it]
|
||||
2026-03-13 14:28:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 14:28:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 14:28:42 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 14:29:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 14:29:17 - evalscope - INFO: Predicting[math_500@Level 1]: 79%| 34/43 [Elapsed: 06:00 < Remaining: 01:39, 11.05s/it]
|
||||
2026-03-13 14:29:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 14:29:38 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 14:30:14 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 14:30:17 - evalscope - INFO: Predicting[math_500@Level 1]: 93%| 40/43 [Elapsed: 07:00 < Remaining: 00:34, 11.62s/it]
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Predicting[math_500@Level 1]: 100%| 43/43 [Elapsed: 07:52 < Remaining: 00:00, 14.61s/it]
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Finished getting predictions for subset: Level 1.
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Getting reviews for subset: Level 1
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Reviewing 43 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Reviewing[math_500@Level 1]: 100%| 43/43 [Elapsed: 00:00 < Remaining: 00:00, 593.02it/s]
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Finished reviewing subset: Level 1. Total reviewed: 43
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Aggregating scores for subset: Level 1
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Evaluating [math_500] 20%| 1/5 [Elapsed: 07:53 < Remaining: 31:32, 473.01s/subset]
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Evaluating subset: Level 2
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Getting predictions for subset: Level 2
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Processing 90 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=1, inflight=1)
|
||||
2026-03-13 14:31:09 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 14:31:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=86, inflight=2)
|
||||
2026-03-13 14:31:59 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=84, inflight=2)
|
||||
2026-03-13 14:32:09 - evalscope - INFO: Predicting[math_500@Level 2]: 3%| 3/90 [Elapsed: 01:00 < Remaining: 38:49, 26.78s/it]
|
||||
2026-03-13 14:32:27 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=82, inflight=2)
|
||||
2026-03-13 14:32:39 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=80, inflight=2)
|
||||
2026-03-13 14:33:09 - evalscope - INFO: Predicting[math_500@Level 2]: 7%| 6/90 [Elapsed: 02:00 < Remaining: 17:50, 12.74s/it]
|
||||
2026-03-13 14:33:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=78, inflight=2)
|
||||
2026-03-13 14:33:41 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=76, inflight=2)
|
||||
2026-03-13 14:33:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=74, inflight=2)
|
||||
2026-03-13 14:34:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=72, inflight=2)
|
||||
2026-03-13 14:34:10 - evalscope - INFO: Predicting[math_500@Level 2]: 16%| 14/90 [Elapsed: 03:00 < Remaining: 14:19, 11.31s/it]
|
||||
2026-03-13 14:34:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=70, inflight=2)
|
||||
2026-03-13 14:34:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=68, inflight=2)
|
||||
2026-03-13 14:35:05 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=66, inflight=2)
|
||||
2026-03-13 14:35:10 - evalscope - INFO: Predicting[math_500@Level 2]: 22%| 20/90 [Elapsed: 04:00 < Remaining: 13:01, 11.16s/it]
|
||||
2026-03-13 14:35:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=64, inflight=2)
|
||||
2026-03-13 14:36:10 - evalscope - INFO: Predicting[math_500@Level 2]: 24%| 22/90 [Elapsed: 05:00 < Remaining: 18:13, 16.08s/it]
|
||||
2026-03-13 14:36:17 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=62, inflight=2)
|
||||
2026-03-13 14:36:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=60, inflight=2)
|
||||
2026-03-13 14:37:06 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=58, inflight=2)
|
||||
2026-03-13 14:37:10 - evalscope - INFO: Predicting[math_500@Level 2]: 31%| 28/90 [Elapsed: 06:00 < Remaining: 13:44, 13.31s/it]
|
||||
2026-03-13 14:37:14 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=56, inflight=2)
|
||||
2026-03-13 14:38:04 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=54, inflight=2)
|
||||
2026-03-13 14:38:10 - evalscope - INFO: Predicting[math_500@Level 2]: 36%| 32/90 [Elapsed: 07:00 < Remaining: 14:13, 14.72s/it]
|
||||
2026-03-13 14:38:30 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=52, inflight=2)
|
||||
2026-03-13 14:38:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=50, inflight=2)
|
||||
2026-03-13 14:39:10 - evalscope - INFO: Predicting[math_500@Level 2]: 40%| 36/90 [Elapsed: 08:00 < Remaining: 10:23, 11.54s/it]
|
||||
2026-03-13 14:39:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=48, inflight=2)
|
||||
2026-03-13 14:39:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=46, inflight=2)
|
||||
2026-03-13 14:39:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=44, inflight=2)
|
||||
2026-03-13 14:39:58 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=42, inflight=2)
|
||||
2026-03-13 14:40:05 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=40, inflight=2)
|
||||
2026-03-13 14:40:10 - evalscope - INFO: Predicting[math_500@Level 2]: 51%| 46/90 [Elapsed: 09:00 < Remaining: 05:39, 7.72s/it]
|
||||
2026-03-13 14:40:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=38, inflight=2)
|
||||
2026-03-13 14:41:10 - evalscope - INFO: Predicting[math_500@Level 2]: 53%| 48/90 [Elapsed: 10:00 < Remaining: 09:08, 13.06s/it]
|
||||
2026-03-13 14:41:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=36, inflight=2)
|
||||
2026-03-13 14:41:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=34, inflight=2)
|
||||
2026-03-13 14:42:10 - evalscope - INFO: Predicting[math_500@Level 2]: 58%| 52/90 [Elapsed: 11:00 < Remaining: 06:35, 10.41s/it]
|
||||
2026-03-13 14:42:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=32, inflight=2)
|
||||
2026-03-13 14:42:25 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=30, inflight=2)
|
||||
2026-03-13 14:43:10 - evalscope - INFO: Predicting[math_500@Level 2]: 62%| 56/90 [Elapsed: 12:00 < Remaining: 06:02, 10.65s/it]
|
||||
2026-03-13 14:43:54 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=28, inflight=2)
|
||||
2026-03-13 14:43:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=26, inflight=2)
|
||||
2026-03-13 14:44:10 - evalscope - INFO: Predicting[math_500@Level 2]: 67%| 60/90 [Elapsed: 13:00 < Remaining: 07:35, 15.17s/it]
|
||||
2026-03-13 14:44:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=24, inflight=2)
|
||||
2026-03-13 14:44:32 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=22, inflight=2)
|
||||
2026-03-13 14:45:10 - evalscope - INFO: Predicting[math_500@Level 2]: 71%| 64/90 [Elapsed: 14:00 < Remaining: 05:01, 11.59s/it]
|
||||
2026-03-13 14:45:49 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=20, inflight=2)
|
||||
2026-03-13 14:45:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=18, inflight=2)
|
||||
2026-03-13 14:46:10 - evalscope - INFO: Predicting[math_500@Level 2]: 76%| 68/90 [Elapsed: 15:00 < Remaining: 05:25, 14.82s/it]
|
||||
2026-03-13 14:46:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=16, inflight=2)
|
||||
2026-03-13 14:46:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=14, inflight=2)
|
||||
2026-03-13 14:47:10 - evalscope - INFO: Predicting[math_500@Level 2]: 80%| 72/90 [Elapsed: 16:00 < Remaining: 03:31, 11.73s/it]
|
||||
2026-03-13 14:47:28 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=12, inflight=2)
|
||||
2026-03-13 14:47:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=10, inflight=2)
|
||||
2026-03-13 14:47:58 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=8, inflight=2)
|
||||
2026-03-13 14:48:10 - evalscope - INFO: Predicting[math_500@Level 2]: 87%| 78/90 [Elapsed: 17:00 < Remaining: 02:14, 11.24s/it]
|
||||
2026-03-13 14:48:26 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=6, inflight=2)
|
||||
2026-03-13 14:48:43 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=4, inflight=2)
|
||||
2026-03-13 14:48:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=2, inflight=2)
|
||||
2026-03-13 14:49:10 - evalscope - INFO: Predicting[math_500@Level 2]: 93%| 84/90 [Elapsed: 18:00 < Remaining: 00:56, 9.43s/it]
|
||||
2026-03-13 14:49:54 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=0, inflight=2)
|
||||
2026-03-13 14:50:10 - evalscope - INFO: Predicting[math_500@Level 2]: 96%| 86/90 [Elapsed: 19:00 < Remaining: 01:02, 15.73s/it]
|
||||
2026-03-13 14:51:10 - evalscope - INFO: Predicting[math_500@Level 2]: 98%| 88/90 [Elapsed: 20:00 < Remaining: 00:26, 13.49s/it]
|
||||
2026-03-13 14:51:19 - evalscope - INFO: Predicting[math_500@Level 2]: 100%| 90/90 [Elapsed: 20:09 < Remaining: 00:00, 19.68s/it]
|
||||
2026-03-13 14:51:19 - evalscope - INFO: Finished getting predictions for subset: Level 2.
|
||||
2026-03-13 14:51:19 - evalscope - INFO: Getting reviews for subset: Level 2
|
||||
2026-03-13 14:51:19 - evalscope - INFO: Reviewing 90 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Reviewing[math_500@Level 2]: 100%| 90/90 [Elapsed: 00:00 < Remaining: 00:00, 223.80it/s]
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Finished reviewing subset: Level 2. Total reviewed: 90
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Aggregating scores for subset: Level 2
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Evaluating [math_500] 40%| 2/5 [Elapsed: 28:03 < Remaining: 45:20, 906.73s/subset]
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Evaluating subset: Level 3
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Getting predictions for subset: Level 3
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Processing 105 samples, if data is large, it may take a while.
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 14:51:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=102, inflight=2)
|
||||
2026-03-13 14:51:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=100, inflight=2)
|
||||
2026-03-13 14:51:55 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=98, inflight=2)
|
||||
2026-03-13 14:52:20 - evalscope - INFO: Predicting[math_500@Level 3]: 3%| 3/105 [Elapsed: 01:00 < Remaining: 31:07, 18.30s/it]
|
||||
2026-03-13 14:52:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=96, inflight=2)
|
||||
2026-03-13 14:53:19 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=94, inflight=2)
|
||||
2026-03-13 14:53:20 - evalscope - INFO: Predicting[math_500@Level 3]: 6%| 6/105 [Elapsed: 02:00 < Remaining: 34:17, 20.79s/it]
|
||||
2026-03-13 14:53:58 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=92, inflight=2)
|
||||
2026-03-13 14:54:20 - evalscope - INFO: Predicting[math_500@Level 3]: 9%| 9/105 [Elapsed: 03:00 < Remaining: 32:06, 20.07s/it]
|
||||
2026-03-13 14:54:40 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=90, inflight=2)
|
||||
2026-03-13 14:54:52 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=88, inflight=2)
|
||||
2026-03-13 14:55:09 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=86, inflight=2)
|
||||
2026-03-13 14:55:20 - evalscope - INFO: Predicting[math_500@Level 3]: 14%| 15/105 [Elapsed: 04:00 < Remaining: 20:01, 13.35s/it]
|
||||
2026-03-13 14:56:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=84, inflight=2)
|
||||
2026-03-13 14:56:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=82, inflight=2)
|
||||
2026-03-13 14:56:20 - evalscope - INFO: Predicting[math_500@Level 3]: 18%| 19/105 [Elapsed: 05:00 < Remaining: 19:08, 13.35s/it]
|
||||
2026-03-13 14:56:42 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=80, inflight=2)
|
||||
2026-03-13 14:57:20 - evalscope - INFO: Predicting[math_500@Level 3]: 20%| 21/105 [Elapsed: 06:00 < Remaining: 19:11, 13.71s/it]
|
||||
2026-03-13 14:57:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=78, inflight=2)
|
||||
2026-03-13 14:58:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=76, inflight=2)
|
||||
2026-03-13 14:58:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=74, inflight=2)
|
||||
2026-03-13 14:58:20 - evalscope - INFO: Predicting[math_500@Level 3]: 25%| 26/105 [Elapsed: 07:00 < Remaining: 17:29, 13.28s/it]
|
||||
2026-03-13 14:58:47 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=72, inflight=2)
|
||||
2026-03-13 14:59:20 - evalscope - INFO: Predicting[math_500@Level 3]: 28%| 29/105 [Elapsed: 08:00 < Remaining: 17:06, 13.50s/it]
|
||||
2026-03-13 14:59:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=70, inflight=2)
|
||||
2026-03-13 15:00:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=68, inflight=2)
|
||||
2026-03-13 15:00:18 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=66, inflight=2)
|
||||
2026-03-13 15:00:20 - evalscope - INFO: Predicting[math_500@Level 3]: 33%| 35/105 [Elapsed: 09:00 < Remaining: 14:45, 12.65s/it]
|
||||
2026-03-13 15:01:20 - evalscope - INFO: Predicting[math_500@Level 3]: 33%| 35/105 [Elapsed: 10:00 < Remaining: 14:45, 12.65s/it]
|
||||
2026-03-13 15:01:25 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=64, inflight=2)
|
||||
2026-03-13 15:01:51 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=62, inflight=2)
|
||||
2026-03-13 15:02:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=60, inflight=2)
|
||||
2026-03-13 15:02:20 - evalscope - INFO: Predicting[math_500@Level 3]: 38%| 40/105 [Elapsed: 11:00 < Remaining: 17:38, 16.28s/it]
|
||||
2026-03-13 15:02:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=58, inflight=2)
|
||||
2026-03-13 15:03:20 - evalscope - INFO: Predicting[math_500@Level 3]: 41%| 43/105 [Elapsed: 12:00 < Remaining: 15:20, 14.85s/it]
|
||||
2026-03-13 15:03:23 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=56, inflight=2)
|
||||
2026-03-13 15:04:15 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=54, inflight=2)
|
||||
2026-03-13 15:04:20 - evalscope - INFO: Predicting[math_500@Level 3]: 45%| 47/105 [Elapsed: 13:00 < Remaining: 18:46, 19.43s/it]
|
||||
2026-03-13 15:04:44 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=52, inflight=2)
|
||||
2026-03-13 15:04:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=50, inflight=2)
|
||||
2026-03-13 15:05:09 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=48, inflight=2)
|
||||
2026-03-13 15:05:19 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=46, inflight=2)
|
||||
2026-03-13 15:05:20 - evalscope - INFO: Predicting[math_500@Level 3]: 52%| 55/105 [Elapsed: 14:00 < Remaining: 08:25, 10.11s/it]
|
||||
2026-03-13 15:05:53 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=44, inflight=2)
|
||||
2026-03-13 15:06:13 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=42, inflight=2)
|
||||
2026-03-13 15:06:20 - evalscope - INFO: Predicting[math_500@Level 3]: 56%| 59/105 [Elapsed: 15:00 < Remaining: 08:50, 11.53s/it]
|
||||
2026-03-13 15:06:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=40, inflight=2)
|
||||
2026-03-13 15:06:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=38, inflight=2)
|
||||
2026-03-13 15:07:20 - evalscope - INFO: Predicting[math_500@Level 3]: 60%| 63/105 [Elapsed: 16:00 < Remaining: 07:35, 10.84s/it]
|
||||
2026-03-13 15:07:39 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=36, inflight=2)
|
||||
2026-03-13 15:08:13 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=34, inflight=2)
|
||||
2026-03-13 15:08:20 - evalscope - INFO: Predicting[math_500@Level 3]: 64%| 67/105 [Elapsed: 17:00 < Remaining: 09:27, 14.93s/it]
|
||||
2026-03-13 15:08:32 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=32, inflight=2)
|
||||
2026-03-13 15:09:20 - evalscope - INFO: Predicting[math_500@Level 3]: 66%| 69/105 [Elapsed: 18:00 < Remaining: 07:58, 13.30s/it]
|
||||
2026-03-13 15:09:44 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=30, inflight=2)
|
||||
2026-03-13 15:10:02 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=28, inflight=2)
|
||||
2026-03-13 15:10:20 - evalscope - INFO: Predicting[math_500@Level 3]: 70%| 73/105 [Elapsed: 19:00 < Remaining: 09:00, 16.89s/it]
|
||||
2026-03-13 15:10:28 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=26, inflight=2)
|
||||
2026-03-13 15:11:10 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=24, inflight=2)
|
||||
2026-03-13 15:11:20 - evalscope - INFO: Predicting[math_500@Level 3]: 73%| 77/105 [Elapsed: 20:00 < Remaining: 08:01, 17.20s/it]
|
||||
2026-03-13 15:11:35 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=22, inflight=2)
|
||||
2026-03-13 15:11:55 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=20, inflight=2)
|
||||
2026-03-13 15:12:20 - evalscope - INFO: Predicting[math_500@Level 3]: 77%| 81/105 [Elapsed: 21:00 < Remaining: 05:37, 14.06s/it]
|
||||
2026-03-13 15:12:55 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=18, inflight=2)
|
||||
2026-03-13 15:13:05 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=16, inflight=2)
|
||||
2026-03-13 15:13:20 - evalscope - INFO: Predicting[math_500@Level 3]: 81%| 85/105 [Elapsed: 22:00 < Remaining: 04:53, 14.69s/it]
|
||||
2026-03-13 15:13:22 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=14, inflight=2)
|
||||
2026-03-13 15:13:38 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=12, inflight=2)
|
||||
2026-03-13 15:14:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=10, inflight=2)
|
||||
2026-03-13 15:14:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=8, inflight=2)
|
||||
2026-03-13 15:14:20 - evalscope - INFO: Predicting[math_500@Level 3]: 88%| 92/105 [Elapsed: 23:00 < Remaining: 02:12, 10.17s/it]
|
||||
2026-03-13 15:14:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=6, inflight=2)
|
||||
2026-03-13 15:14:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=4, inflight=2)
|
||||
2026-03-13 15:15:20 - evalscope - INFO: Predicting[math_500@Level 3]: 92%| 97/105 [Elapsed: 24:00 < Remaining: 01:17, 9.73s/it]
|
||||
2026-03-13 15:15:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=2, inflight=2)
|
||||
2026-03-13 15:16:20 - evalscope - INFO: Predicting[math_500@Level 3]: 94%| 99/105 [Elapsed: 25:00 < Remaining: 01:29, 14.91s/it]
|
||||
2026-03-13 15:16:33 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=0, inflight=2)
|
||||
2026-03-13 15:17:20 - evalscope - INFO: Predicting[math_500@Level 3]: 96%| 101/105 [Elapsed: 26:00 < Remaining: 01:06, 16.74s/it]
|
||||
2026-03-13 15:17:24 - evalscope - INFO: Predicting[math_500@Level 3]: 100%| 105/105 [Elapsed: 26:04 < Remaining: 00:00, 19.37s/it]
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Finished getting predictions for subset: Level 3.
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Getting reviews for subset: Level 3
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Reviewing 105 samples, if data is large, it may take a while.
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Reviewing[math_500@Level 3]: 100%| 105/105 [Elapsed: 00:00 < Remaining: 00:00, 106.24it/s]
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Finished reviewing subset: Level 3. Total reviewed: 105
|
||||
2026-03-13 15:17:25 - evalscope - INFO: Aggregating scores for subset: Level 3
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Evaluating [math_500] 60%| 3/5 [Elapsed: 54:09 < Remaining: 40:15, 1207.67s/subset]
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Evaluating subset: Level 4
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Getting predictions for subset: Level 4
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Processing 128 samples, if data is large, it may take a while.
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 15:17:26 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=116, inflight=2)
|
||||
2026-03-13 15:17:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=123, inflight=2)
|
||||
2026-03-13 15:18:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=121, inflight=2)
|
||||
2026-03-13 15:18:26 - evalscope - INFO: Predicting[math_500@Level 4]: 2%| 3/128 [Elapsed: 01:00 < Remaining: 27:38, 13.27s/it]
|
||||
2026-03-13 15:19:26 - evalscope - INFO: Predicting[math_500@Level 4]: 2%| 3/128 [Elapsed: 02:00 < Remaining: 27:38, 13.27s/it]
|
||||
2026-03-13 15:19:35 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=119, inflight=2)
|
||||
2026-03-13 15:19:44 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=117, inflight=2)
|
||||
2026-03-13 15:20:03 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=115, inflight=2)
|
||||
2026-03-13 15:20:26 - evalscope - INFO: Predicting[math_500@Level 4]: 7%| 9/128 [Elapsed: 03:00 < Remaining: 32:26, 16.36s/it]
|
||||
2026-03-13 15:21:15 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=113, inflight=2)
|
||||
2026-03-13 15:21:26 - evalscope - INFO: Predicting[math_500@Level 4]: 9%| 11/128 [Elapsed: 04:00 < Remaining: 46:10, 23.68s/it]
|
||||
2026-03-13 15:21:40 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=111, inflight=2)
|
||||
2026-03-13 15:22:26 - evalscope - INFO: Predicting[math_500@Level 4]: 10%| 13/128 [Elapsed: 05:00 < Remaining: 37:57, 19.80s/it]
|
||||
2026-03-13 15:22:47 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=109, inflight=2)
|
||||
2026-03-13 15:23:17 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=107, inflight=2)
|
||||
2026-03-13 15:23:26 - evalscope - INFO: Predicting[math_500@Level 4]: 13%| 17/128 [Elapsed: 06:00 < Remaining: 39:34, 21.39s/it]
|
||||
2026-03-13 15:23:46 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=105, inflight=2)
|
||||
2026-03-13 15:24:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=103, inflight=2)
|
||||
2026-03-13 15:24:26 - evalscope - INFO: Predicting[math_500@Level 4]: 16%| 21/128 [Elapsed: 07:00 < Remaining: 32:30, 18.23s/it]
|
||||
2026-03-13 15:24:47 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=101, inflight=2)
|
||||
2026-03-13 15:25:26 - evalscope - INFO: Predicting[math_500@Level 4]: 18%| 23/128 [Elapsed: 08:00 < Remaining: 29:38, 16.93s/it]
|
||||
2026-03-13 15:25:26 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=99, inflight=2)
|
||||
2026-03-13 15:25:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=97, inflight=2)
|
||||
2026-03-13 15:26:26 - evalscope - INFO: Predicting[math_500@Level 4]: 21%| 27/128 [Elapsed: 09:00 < Remaining: 28:37, 17.00s/it]
|
||||
2026-03-13 15:26:51 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=95, inflight=2)
|
||||
2026-03-13 15:26:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=93, inflight=2)
|
||||
2026-03-13 15:27:26 - evalscope - INFO: Predicting[math_500@Level 4]: 24%| 31/128 [Elapsed: 10:00 < Remaining: 24:00, 14.85s/it]
|
||||
2026-03-13 15:27:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=91, inflight=2)
|
||||
2026-03-13 15:28:16 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=89, inflight=2)
|
||||
2026-03-13 15:28:26 - evalscope - INFO: Predicting[math_500@Level 4]: 27%| 35/128 [Elapsed: 11:00 < Remaining: 27:23, 17.67s/it]
|
||||
2026-03-13 15:28:29 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=87, inflight=2)
|
||||
2026-03-13 15:28:36 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=85, inflight=2)
|
||||
2026-03-13 15:29:06 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=83, inflight=2)
|
||||
2026-03-13 15:29:26 - evalscope - INFO: Predicting[math_500@Level 4]: 32%| 41/128 [Elapsed: 12:00 < Remaining: 17:45, 12.25s/it]
|
||||
2026-03-13 15:29:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=81, inflight=2)
|
||||
2026-03-13 15:30:26 - evalscope - INFO: Predicting[math_500@Level 4]: 34%| 43/128 [Elapsed: 13:00 < Remaining: 23:25, 16.53s/it]
|
||||
2026-03-13 15:30:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=79, inflight=2)
|
||||
2026-03-13 15:30:48 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=77, inflight=2)
|
||||
2026-03-13 15:31:26 - evalscope - INFO: Predicting[math_500@Level 4]: 37%| 47/128 [Elapsed: 14:00 < Remaining: 18:22, 13.61s/it]
|
||||
2026-03-13 15:32:13 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=75, inflight=2)
|
||||
2026-03-13 15:32:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=73, inflight=2)
|
||||
2026-03-13 15:32:26 - evalscope - INFO: Predicting[math_500@Level 4]: 40%| 51/128 [Elapsed: 15:00 < Remaining: 21:13, 16.54s/it]
|
||||
2026-03-13 15:33:26 - evalscope - INFO: Predicting[math_500@Level 4]: 40%| 51/128 [Elapsed: 16:00 < Remaining: 21:13, 16.54s/it]
|
||||
2026-03-13 15:33:50 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=71, inflight=2)
|
||||
2026-03-13 15:33:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=69, inflight=2)
|
||||
2026-03-13 15:34:17 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=67, inflight=2)
|
||||
2026-03-13 15:34:26 - evalscope - INFO: Predicting[math_500@Level 4]: 45%| 57/128 [Elapsed: 17:00 < Remaining: 19:05, 16.13s/it]
|
||||
2026-03-13 15:35:01 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=65, inflight=2)
|
||||
2026-03-13 15:35:07 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=63, inflight=2)
|
||||
2026-03-13 15:35:26 - evalscope - INFO: Predicting[math_500@Level 4]: 48%| 61/128 [Elapsed: 18:00 < Remaining: 14:54, 13.34s/it]
|
||||
2026-03-13 15:35:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=61, inflight=2)
|
||||
2026-03-13 15:36:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=59, inflight=2)
|
||||
2026-03-13 15:36:26 - evalscope - INFO: Predicting[math_500@Level 4]: 51%| 65/128 [Elapsed: 19:00 < Remaining: 14:26, 13.76s/it]
|
||||
2026-03-13 15:36:31 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=57, inflight=2)
|
||||
2026-03-13 15:37:04 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=55, inflight=2)
|
||||
2026-03-13 15:37:19 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=53, inflight=2)
|
||||
2026-03-13 15:37:26 - evalscope - INFO: Predicting[math_500@Level 4]: 55%| 71/128 [Elapsed: 20:00 < Remaining: 11:31, 12.13s/it]
|
||||
2026-03-13 15:37:43 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=51, inflight=2)
|
||||
2026-03-13 15:38:26 - evalscope - INFO: Predicting[math_500@Level 4]: 57%| 73/128 [Elapsed: 21:00 < Remaining: 11:05, 12.09s/it]
|
||||
2026-03-13 15:38:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=49, inflight=2)
|
||||
2026-03-13 15:39:21 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=47, inflight=2)
|
||||
2026-03-13 15:39:26 - evalscope - INFO: Predicting[math_500@Level 4]: 60%| 77/128 [Elapsed: 22:00 < Remaining: 14:44, 17.34s/it]
|
||||
2026-03-13 15:39:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=45, inflight=2)
|
||||
2026-03-13 15:40:11 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=43, inflight=2)
|
||||
2026-03-13 15:40:26 - evalscope - INFO: Predicting[math_500@Level 4]: 63%| 81/128 [Elapsed: 23:00 < Remaining: 12:10, 15.55s/it]
|
||||
2026-03-13 15:40:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=41, inflight=2)
|
||||
2026-03-13 15:40:45 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 15:41:26 - evalscope - INFO: Predicting[math_500@Level 4]: 66%| 85/128 [Elapsed: 24:00 < Remaining: 08:14, 11.51s/it]
|
||||
2026-03-13 15:41:52 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 15:42:15 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 15:42:27 - evalscope - INFO: Predicting[math_500@Level 4]: 70%| 89/128 [Elapsed: 25:00 < Remaining: 10:28, 16.13s/it]
|
||||
2026-03-13 15:43:24 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 15:43:27 - evalscope - INFO: Predicting[math_500@Level 4]: 71%| 91/128 [Elapsed: 26:00 < Remaining: 13:20, 21.64s/it]
|
||||
2026-03-13 15:43:47 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 15:44:27 - evalscope - INFO: Predicting[math_500@Level 4]: 73%| 93/128 [Elapsed: 27:00 < Remaining: 10:51, 18.60s/it]
|
||||
2026-03-13 15:44:44 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 15:44:55 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 15:45:27 - evalscope - INFO: Predicting[math_500@Level 4]: 76%| 97/128 [Elapsed: 28:01 < Remaining: 08:39, 16.75s/it]
|
||||
2026-03-13 15:46:20 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 15:46:27 - evalscope - INFO: Predicting[math_500@Level 4]: 77%| 99/128 [Elapsed: 29:01 < Remaining: 11:49, 24.48s/it]
|
||||
2026-03-13 15:46:31 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 15:46:57 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 15:47:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 15:47:27 - evalscope - INFO: Predicting[math_500@Level 4]: 82%| 105/128 [Elapsed: 30:01 < Remaining: 06:07, 15.99s/it]
|
||||
2026-03-13 15:47:51 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 15:48:27 - evalscope - INFO: Predicting[math_500@Level 4]: 84%| 107/128 [Elapsed: 31:01 < Remaining: 05:20, 15.24s/it]
|
||||
2026-03-13 15:48:28 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 15:48:55 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 15:49:27 - evalscope - INFO: Predicting[math_500@Level 4]: 87%| 111/128 [Elapsed: 32:01 < Remaining: 04:22, 15.45s/it]
|
||||
2026-03-13 15:49:58 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 15:50:27 - evalscope - INFO: Predicting[math_500@Level 4]: 88%| 113/128 [Elapsed: 33:01 < Remaining: 05:04, 20.27s/it]
|
||||
2026-03-13 15:50:31 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 15:50:52 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 15:51:27 - evalscope - INFO: Predicting[math_500@Level 4]: 91%| 117/128 [Elapsed: 34:01 < Remaining: 03:02, 16.55s/it]
|
||||
2026-03-13 15:51:30 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 15:52:23 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 15:52:27 - evalscope - INFO: Predicting[math_500@Level 4]: 95%| 121/128 [Elapsed: 35:01 < Remaining: 02:20, 20.05s/it]
|
||||
2026-03-13 15:53:00 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 15:53:05 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 15:53:27 - evalscope - INFO: Predicting[math_500@Level 4]: 98%| 125/128 [Elapsed: 36:01 < Remaining: 00:43, 14.46s/it]
|
||||
2026-03-13 15:54:14 - evalscope - INFO: Predicting[math_500@Level 4]: 100%| 128/128 [Elapsed: 36:48 < Remaining: 00:00, 15.97s/it]
|
||||
2026-03-13 15:54:14 - evalscope - INFO: Finished getting predictions for subset: Level 4.
|
||||
2026-03-13 15:54:14 - evalscope - INFO: Getting reviews for subset: Level 4
|
||||
2026-03-13 15:54:14 - evalscope - INFO: Reviewing 128 samples, if data is large, it may take a while.
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Reviewing[math_500@Level 4]: 100%| 128/128 [Elapsed: 00:01 < Remaining: 00:00, 1.01s/it]
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Finished reviewing subset: Level 4. Total reviewed: 128
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Aggregating scores for subset: Level 4
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Evaluating [math_500] 80%| 4/5 [Elapsed: 1:30:59 < Remaining: 26:43, 1603.43s/subset]
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Evaluating subset: Level 5
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Getting predictions for subset: Level 5
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Processing 134 samples, if data is large, it may take a while.
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Dispatcher: Worker-0 <- 1 prompts (pending=0, inflight=1)
|
||||
2026-03-13 15:54:16 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=116, inflight=2)
|
||||
2026-03-13 15:55:16 - evalscope - INFO: Predicting[math_500@Level 5]: 0%| 0/134 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-13 15:55:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=129, inflight=2)
|
||||
2026-03-13 15:55:53 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=127, inflight=2)
|
||||
2026-03-13 15:56:16 - evalscope - INFO: Predicting[math_500@Level 5]: 2%| 3/134 [Elapsed: 02:00 < Remaining: 1:31:14, 41.79s/it]
|
||||
2026-03-13 15:57:16 - evalscope - INFO: Predicting[math_500@Level 5]: 2%| 3/134 [Elapsed: 03:00 < Remaining: 1:31:14, 41.79s/it]
|
||||
2026-03-13 15:57:18 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=125, inflight=2)
|
||||
2026-03-13 15:57:24 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=123, inflight=2)
|
||||
2026-03-13 15:58:16 - evalscope - INFO: Predicting[math_500@Level 5]: 5%| 7/134 [Elapsed: 04:00 < Remaining: 50:10, 23.71s/it]
|
||||
2026-03-13 15:58:56 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=121, inflight=2)
|
||||
2026-03-13 15:58:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=119, inflight=2)
|
||||
2026-03-13 15:59:16 - evalscope - INFO: Predicting[math_500@Level 5]: 8%| 11/134 [Elapsed: 05:00 < Remaining: 43:24, 21.17s/it]
|
||||
2026-03-13 15:59:32 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=117, inflight=2)
|
||||
2026-03-13 16:00:04 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=115, inflight=2)
|
||||
2026-03-13 16:00:16 - evalscope - INFO: Predicting[math_500@Level 5]: 11%| 15/134 [Elapsed: 06:00 < Remaining: 36:29, 18.40s/it]
|
||||
2026-03-13 16:00:57 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=113, inflight=2)
|
||||
2026-03-13 16:01:08 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=111, inflight=2)
|
||||
2026-03-13 16:01:16 - evalscope - INFO: Predicting[math_500@Level 5]: 14%| 19/134 [Elapsed: 07:00 < Remaining: 30:56, 16.14s/it]
|
||||
2026-03-13 16:02:16 - evalscope - INFO: Predicting[math_500@Level 5]: 14%| 19/134 [Elapsed: 08:00 < Remaining: 30:56, 16.14s/it]
|
||||
2026-03-13 16:02:34 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=109, inflight=2)
|
||||
2026-03-13 16:02:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=107, inflight=2)
|
||||
2026-03-13 16:03:16 - evalscope - INFO: Predicting[math_500@Level 5]: 17%| 23/134 [Elapsed: 09:00 < Remaining: 35:03, 18.95s/it]
|
||||
2026-03-13 16:04:09 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=105, inflight=2)
|
||||
2026-03-13 16:04:16 - evalscope - INFO: Predicting[math_500@Level 5]: 19%| 25/134 [Elapsed: 10:00 < Remaining: 46:38, 25.67s/it]
|
||||
2026-03-13 16:04:22 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=103, inflight=2)
|
||||
2026-03-13 16:05:16 - evalscope - INFO: Predicting[math_500@Level 5]: 20%| 27/134 [Elapsed: 11:00 < Remaining: 35:25, 19.86s/it]
|
||||
2026-03-13 16:05:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=101, inflight=2)
|
||||
2026-03-13 16:05:44 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=99, inflight=2)
|
||||
2026-03-13 16:06:16 - evalscope - INFO: Predicting[math_500@Level 5]: 23%| 31/134 [Elapsed: 12:00 < Remaining: 33:17, 19.40s/it]
|
||||
2026-03-13 16:06:55 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=97, inflight=2)
|
||||
2026-03-13 16:07:16 - evalscope - INFO: Predicting[math_500@Level 5]: 25%| 33/134 [Elapsed: 13:00 < Remaining: 40:33, 24.10s/it]
|
||||
2026-03-13 16:07:20 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=95, inflight=2)
|
||||
2026-03-13 16:08:16 - evalscope - INFO: Predicting[math_500@Level 5]: 26%| 35/134 [Elapsed: 14:00 < Remaining: 34:00, 20.61s/it]
|
||||
2026-03-13 16:08:33 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=93, inflight=2)
|
||||
2026-03-13 16:08:51 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=91, inflight=2)
|
||||
2026-03-13 16:09:16 - evalscope - INFO: Predicting[math_500@Level 5]: 29%| 39/134 [Elapsed: 15:00 < Remaining: 32:24, 20.47s/it]
|
||||
2026-03-13 16:09:53 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=89, inflight=2)
|
||||
2026-03-13 16:10:16 - evalscope - INFO: Predicting[math_500@Level 5]: 31%| 41/134 [Elapsed: 16:00 < Remaining: 36:38, 23.64s/it]
|
||||
2026-03-13 16:10:27 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=87, inflight=2)
|
||||
2026-03-13 16:11:16 - evalscope - INFO: Predicting[math_500@Level 5]: 32%| 43/134 [Elapsed: 17:00 < Remaining: 32:49, 21.65s/it]
|
||||
2026-03-13 16:11:19 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=85, inflight=2)
|
||||
2026-03-13 16:12:04 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=83, inflight=2)
|
||||
2026-03-13 16:12:16 - evalscope - INFO: Predicting[math_500@Level 5]: 35%| 47/134 [Elapsed: 18:00 < Remaining: 33:18, 22.97s/it]
|
||||
2026-03-13 16:12:26 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=81, inflight=2)
|
||||
2026-03-13 16:13:16 - evalscope - INFO: Predicting[math_500@Level 5]: 37%| 49/134 [Elapsed: 19:00 < Remaining: 27:14, 19.23s/it]
|
||||
2026-03-13 16:13:41 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=79, inflight=2)
|
||||
2026-03-13 16:14:04 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=77, inflight=2)
|
||||
2026-03-13 16:14:16 - evalscope - INFO: Predicting[math_500@Level 5]: 40%| 53/134 [Elapsed: 20:00 < Remaining: 28:01, 20.75s/it]
|
||||
2026-03-13 16:15:05 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=75, inflight=2)
|
||||
2026-03-13 16:15:16 - evalscope - INFO: Predicting[math_500@Level 5]: 41%| 55/134 [Elapsed: 21:00 < Remaining: 31:10, 23.68s/it]
|
||||
2026-03-13 16:15:35 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=73, inflight=2)
|
||||
2026-03-13 16:16:15 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=71, inflight=2)
|
||||
2026-03-13 16:16:16 - evalscope - INFO: Predicting[math_500@Level 5]: 44%| 59/134 [Elapsed: 22:00 < Remaining: 25:56, 20.76s/it]
|
||||
2026-03-13 16:16:46 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=69, inflight=2)
|
||||
2026-03-13 16:17:16 - evalscope - INFO: Predicting[math_500@Level 5]: 46%| 61/134 [Elapsed: 23:00 < Remaining: 23:20, 19.18s/it]
|
||||
2026-03-13 16:17:50 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=67, inflight=2)
|
||||
2026-03-13 16:18:16 - evalscope - INFO: Predicting[math_500@Level 5]: 47%| 63/134 [Elapsed: 24:00 < Remaining: 27:15, 23.03s/it]
|
||||
2026-03-13 16:18:24 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=65, inflight=2)
|
||||
2026-03-13 16:18:45 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=63, inflight=2)
|
||||
2026-03-13 16:19:16 - evalscope - INFO: Predicting[math_500@Level 5]: 50%| 67/134 [Elapsed: 25:00 < Remaining: 20:03, 17.96s/it]
|
||||
2026-03-13 16:19:57 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=61, inflight=2)
|
||||
2026-03-13 16:20:12 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=59, inflight=2)
|
||||
2026-03-13 16:20:16 - evalscope - INFO: Predicting[math_500@Level 5]: 53%| 71/134 [Elapsed: 26:00 < Remaining: 19:32, 18.62s/it]
|
||||
2026-03-13 16:20:39 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=57, inflight=2)
|
||||
2026-03-13 16:21:17 - evalscope - INFO: Predicting[math_500@Level 5]: 54%| 73/134 [Elapsed: 27:00 < Remaining: 17:22, 17.08s/it]
|
||||
2026-03-13 16:21:45 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=55, inflight=2)
|
||||
2026-03-13 16:22:09 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=53, inflight=2)
|
||||
2026-03-13 16:22:17 - evalscope - INFO: Predicting[math_500@Level 5]: 57%| 77/134 [Elapsed: 28:00 < Remaining: 17:57, 18.90s/it]
|
||||
2026-03-13 16:22:59 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=51, inflight=2)
|
||||
2026-03-13 16:23:17 - evalscope - INFO: Predicting[math_500@Level 5]: 59%| 79/134 [Elapsed: 29:00 < Remaining: 19:00, 20.74s/it]
|
||||
2026-03-13 16:23:41 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=49, inflight=2)
|
||||
2026-03-13 16:24:17 - evalscope - INFO: Predicting[math_500@Level 5]: 60%| 81/134 [Elapsed: 30:00 < Remaining: 18:23, 20.82s/it]
|
||||
2026-03-13 16:24:35 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=47, inflight=2)
|
||||
2026-03-13 16:25:12 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=45, inflight=2)
|
||||
2026-03-13 16:25:17 - evalscope - INFO: Predicting[math_500@Level 5]: 63%| 85/134 [Elapsed: 31:00 < Remaining: 17:29, 21.42s/it]
|
||||
2026-03-13 16:26:07 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=43, inflight=2)
|
||||
2026-03-13 16:26:17 - evalscope - INFO: Predicting[math_500@Level 5]: 65%| 87/134 [Elapsed: 32:00 < Remaining: 18:12, 23.25s/it]
|
||||
2026-03-13 16:26:45 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=41, inflight=2)
|
||||
2026-03-13 16:27:17 - evalscope - INFO: Predicting[math_500@Level 5]: 66%| 89/134 [Elapsed: 33:00 < Remaining: 16:28, 21.98s/it]
|
||||
2026-03-13 16:27:38 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=39, inflight=2)
|
||||
2026-03-13 16:28:17 - evalscope - INFO: Predicting[math_500@Level 5]: 68%| 91/134 [Elapsed: 34:00 < Remaining: 16:43, 23.34s/it]
|
||||
2026-03-13 16:28:17 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=37, inflight=2)
|
||||
2026-03-13 16:29:09 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=35, inflight=2)
|
||||
2026-03-13 16:29:17 - evalscope - INFO: Predicting[math_500@Level 5]: 71%| 95/134 [Elapsed: 35:00 < Remaining: 15:10, 23.33s/it]
|
||||
2026-03-13 16:29:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=33, inflight=2)
|
||||
2026-03-13 16:30:17 - evalscope - INFO: Predicting[math_500@Level 5]: 72%| 97/134 [Elapsed: 36:01 < Remaining: 13:40, 22.19s/it]
|
||||
2026-03-13 16:30:39 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=31, inflight=2)
|
||||
2026-03-13 16:31:14 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=29, inflight=2)
|
||||
2026-03-13 16:31:17 - evalscope - INFO: Predicting[math_500@Level 5]: 75%| 101/134 [Elapsed: 37:01 < Remaining: 11:48, 21.48s/it]
|
||||
2026-03-13 16:31:56 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=27, inflight=2)
|
||||
2026-03-13 16:32:17 - evalscope - INFO: Predicting[math_500@Level 5]: 77%| 103/134 [Elapsed: 38:01 < Remaining: 11:01, 21.34s/it]
|
||||
2026-03-13 16:32:48 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=25, inflight=2)
|
||||
2026-03-13 16:33:17 - evalscope - INFO: Predicting[math_500@Level 5]: 78%| 105/134 [Elapsed: 39:01 < Remaining: 10:55, 22.59s/it]
|
||||
2026-03-13 16:33:33 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=23, inflight=2)
|
||||
2026-03-13 16:34:17 - evalscope - INFO: Predicting[math_500@Level 5]: 80%| 107/134 [Elapsed: 40:01 < Remaining: 10:13, 22.71s/it]
|
||||
2026-03-13 16:34:25 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=21, inflight=2)
|
||||
2026-03-13 16:35:05 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=19, inflight=2)
|
||||
2026-03-13 16:35:17 - evalscope - INFO: Predicting[math_500@Level 5]: 83%| 111/134 [Elapsed: 41:01 < Remaining: 08:37, 22.49s/it]
|
||||
2026-03-13 16:35:37 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=17, inflight=2)
|
||||
2026-03-13 16:36:17 - evalscope - INFO: Predicting[math_500@Level 5]: 84%| 113/134 [Elapsed: 42:01 < Remaining: 07:14, 20.69s/it]
|
||||
2026-03-13 16:36:35 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=15, inflight=2)
|
||||
2026-03-13 16:37:09 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=13, inflight=2)
|
||||
2026-03-13 16:37:17 - evalscope - INFO: Predicting[math_500@Level 5]: 87%| 117/134 [Elapsed: 43:01 < Remaining: 06:02, 21.33s/it]
|
||||
2026-03-13 16:38:08 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=11, inflight=2)
|
||||
2026-03-13 16:38:17 - evalscope - INFO: Predicting[math_500@Level 5]: 89%| 119/134 [Elapsed: 44:01 < Remaining: 05:56, 23.79s/it]
|
||||
2026-03-13 16:38:47 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=9, inflight=2)
|
||||
2026-03-13 16:39:17 - evalscope - INFO: Predicting[math_500@Level 5]: 90%| 121/134 [Elapsed: 45:01 < Remaining: 04:52, 22.50s/it]
|
||||
2026-03-13 16:39:46 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=7, inflight=2)
|
||||
2026-03-13 16:40:17 - evalscope - INFO: Predicting[math_500@Level 5]: 92%| 123/134 [Elapsed: 46:01 < Remaining: 04:30, 24.60s/it]
|
||||
2026-03-13 16:40:19 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=5, inflight=2)
|
||||
2026-03-13 16:41:17 - evalscope - INFO: Predicting[math_500@Level 5]: 93%| 125/134 [Elapsed: 47:01 < Remaining: 03:19, 22.17s/it]
|
||||
2026-03-13 16:41:23 - evalscope - INFO: Dispatcher: Worker-1 <- 2 prompts (pending=3, inflight=2)
|
||||
2026-03-13 16:41:49 - evalscope - INFO: Dispatcher: Worker-0 <- 2 prompts (pending=1, inflight=2)
|
||||
2026-03-13 16:42:17 - evalscope - INFO: Predicting[math_500@Level 5]: 96%| 129/134 [Elapsed: 48:01 < Remaining: 01:47, 21.49s/it]
|
||||
2026-03-13 16:42:48 - evalscope - INFO: Dispatcher: Worker-1 <- 1 prompts (pending=0, inflight=2)
|
||||
2026-03-13 16:43:17 - evalscope - INFO: Predicting[math_500@Level 5]: 98%| 131/134 [Elapsed: 49:01 < Remaining: 01:11, 23.74s/it]
|
||||
2026-03-13 16:44:17 - evalscope - INFO: Predicting[math_500@Level 5]: 100%| 134/134 [Elapsed: 50:01 < Remaining: 00:00, 23.17s/it]
|
||||
2026-03-13 16:44:17 - evalscope - INFO: Finished getting predictions for subset: Level 5.
|
||||
2026-03-13 16:44:17 - evalscope - INFO: Getting reviews for subset: Level 5
|
||||
2026-03-13 16:44:17 - evalscope - INFO: Reviewing 134 samples, if data is large, it may take a while.
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Reviewing[math_500@Level 5]: 100%| 134/134 [Elapsed: 00:02 < Remaining: 00:00, 36.68it/s]
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Finished reviewing subset: Level 5. Total reviewed: 134
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Aggregating scores for subset: Level 5
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Evaluating [math_500] 100%| 5/5 [Elapsed: 2:21:02 < Remaining: 00:00, 2108.24s/subset]
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Generating report...
|
||||
2026-03-13 16:44:19 - evalscope - INFO:
|
||||
math_500 report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 1 | 43 | 0.7442 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 2 | 90 | 0.7222 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 3 | 105 | 0.5714 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 4 | 128 | 0.375 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 5 | 134 | 0.2015 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | OVERALL | 500 | 0.464 | - |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500/20260313_142308/reports/llama3_3b_instruct_vallina_full_sft_30k/math_500.json
|
||||
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Benchmark math_500 evaluation finished.
|
||||
2026-03-13 16:44:19 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+===========+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 1 | 43 | 0.7442 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 2 | 90 | 0.7222 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 3 | 105 | 0.5714 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 4 | 128 | 0.375 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | Level 5 | 134 | 0.2015 | default |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | math_500 | mean_acc | OVERALL | 500 | 0.464 | - |
|
||||
+-----------------------------------------+-----------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['math_500']
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500/20260313_142308
|
||||
2026-03-13 16:44:21 - evalscope - INFO: [进度条] MATH500 评测完成 ✓
|
||||
2026-03-13 16:44:21 - evalscope - INFO: [断点续传] MATH500 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MATH500_result.json
|
||||
2026-03-13 16:44:21 - evalscope - INFO: 完成评测 MATH500 (3/8)
|
||||
2026-03-13 16:44:21 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-13 16:44:21 - evalscope - INFO: 正在评估 AMC (repeat: 1次) (剩余: 4个)
|
||||
2026-03-13 16:44:21 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-13 16:44:21 - evalscope - INFO: 开始创建 benchmark AMC 的 TaskConfig
|
||||
2026-03-13 16:44:21 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-13 16:44:21 - evalscope - INFO: [AMC] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['amc'], eval_batch_size=2048
|
||||
2026-03-13 16:44:21 - evalscope - INFO: 开始评测 AMC...
|
||||
2026-03-13 16:44:21 - evalscope - INFO: [进度条] AMC 开始评测
|
||||
2026-03-13 16:44:21 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,54 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@math_500",
|
||||
"dataset_name": "math_500",
|
||||
"dataset_pretty_name": "MATH-500",
|
||||
"dataset_description": "MATH-500 is a benchmark for evaluating mathematical reasoning capabilities of AI models. It consists of 500 diverse math problems across five levels of difficulty, designed to test a model's ability to solve complex mathematical problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.464,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.464,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.5229,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "Level 1",
|
||||
"score": 0.7442,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "Level 2",
|
||||
"score": 0.7222,
|
||||
"num": 90
|
||||
},
|
||||
{
|
||||
"name": "Level 3",
|
||||
"score": 0.5714,
|
||||
"num": 105
|
||||
},
|
||||
{
|
||||
"name": "Level 4",
|
||||
"score": 0.375,
|
||||
"num": 128
|
||||
},
|
||||
{
|
||||
"name": "Level 5",
|
||||
"score": 0.2015,
|
||||
"num": 134
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
56
evalscope/version_20260313_110435/run_1/MATH500_result.json
Normal file
56
evalscope/version_20260313_110435/run_1/MATH500_result.json
Normal file
@@ -0,0 +1,56 @@
|
||||
{
|
||||
"math_500": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@math_500",
|
||||
"dataset_name": "math_500",
|
||||
"dataset_pretty_name": "MATH-500",
|
||||
"dataset_description": "MATH-500 is a benchmark for evaluating mathematical reasoning capabilities of AI models. It consists of 500 diverse math problems across five levels of difficulty, designed to test a model's ability to solve complex mathematical problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.464,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.464,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.5229,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "Level 1",
|
||||
"score": 0.7442,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "Level 2",
|
||||
"score": 0.7222,
|
||||
"num": 90
|
||||
},
|
||||
{
|
||||
"name": "Level 3",
|
||||
"score": 0.5714,
|
||||
"num": 105
|
||||
},
|
||||
{
|
||||
"name": "Level 4",
|
||||
"score": 0.375,
|
||||
"num": 128
|
||||
},
|
||||
{
|
||||
"name": "Level 5",
|
||||
"score": 0.2015,
|
||||
"num": 134
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
mmlu_pro_no_math:
|
||||
aggregation: mean
|
||||
dataset_id: TIGER-Lab/MMLU-Pro
|
||||
default_subset: default
|
||||
description: MMLU-Pro benchmark excluding the math category.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: 'The following are multiple choice questions (with answers)
|
||||
about {subject}. Think step by step and then finish your answer with ''ANSWER:
|
||||
[LETTER]'' (without quotes) where [LETTER] is the correct letter choice.
|
||||
|
||||
|
||||
{examples}
|
||||
|
||||
Answer the following multiple choice question. The last line of your response
|
||||
should be of the following format: ''ANSWER: [LETTER]'' (without quotes) where
|
||||
[LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
Question:
|
||||
|
||||
{question}
|
||||
|
||||
Options:
|
||||
|
||||
{choices}
|
||||
|
||||
'
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: mmlu_pro_no_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: MMLU-Pro (No Math)
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
Question:
|
||||
|
||||
{question}
|
||||
|
||||
Options:
|
||||
|
||||
{choices}
|
||||
|
||||
'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- MCQ
|
||||
- Knowledge
|
||||
train_split: validation
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- mmlu_pro_no_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052
|
||||
@@ -0,0 +1,105 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
mmlu_pro_no_math:
|
||||
aggregation: mean
|
||||
dataset_id: TIGER-Lab/MMLU-Pro
|
||||
default_subset: default
|
||||
description: MMLU-Pro benchmark excluding the math category.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: 'The following are multiple choice questions (with answers)
|
||||
about {subject}. Think step by step and then finish your answer with ''ANSWER:
|
||||
[LETTER]'' (without quotes) where [LETTER] is the correct letter choice.
|
||||
|
||||
|
||||
{examples}
|
||||
|
||||
Answer the following multiple choice question. The last line of your response
|
||||
should be of the following format: ''ANSWER: [LETTER]'' (without quotes) where
|
||||
[LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
Question:
|
||||
|
||||
{question}
|
||||
|
||||
Options:
|
||||
|
||||
{choices}
|
||||
|
||||
'
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc
|
||||
name: mmlu_pro_no_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: MMLU-Pro (No Math)
|
||||
prompt_template: 'Answer the following multiple choice question. The last line
|
||||
of your response should be of the following format: ''ANSWER: [LETTER]'' (without
|
||||
quotes) where [LETTER] is one of {letters}. Think step by step before answering.
|
||||
|
||||
|
||||
Question:
|
||||
|
||||
{question}
|
||||
|
||||
Options:
|
||||
|
||||
{choices}
|
||||
|
||||
'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: null
|
||||
tags:
|
||||
- MCQ
|
||||
- Knowledge
|
||||
train_split: validation
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- mmlu_pro_no_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052
|
||||
@@ -0,0 +1,260 @@
|
||||
2026-03-14 18:00:51 - evalscope - INFO: Running with native backend
|
||||
2026-03-14 18:00:51 - evalscope - INFO: Dump task config to /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052/configs/task_config_5c078b.yaml
|
||||
2026-03-14 18:00:51 - evalscope - INFO: {
|
||||
"model": "VLLMOfflineModelAPI",
|
||||
"model_id": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"model_args": {},
|
||||
"model_task": "text_generation",
|
||||
"chat_template": null,
|
||||
"datasets": [
|
||||
"mmlu_pro_no_math"
|
||||
],
|
||||
"dataset_args": {
|
||||
"mmlu_pro_no_math": {
|
||||
"name": "mmlu_pro_no_math",
|
||||
"dataset_id": "TIGER-Lab/MMLU-Pro",
|
||||
"output_types": [
|
||||
"generation"
|
||||
],
|
||||
"subset_list": [
|
||||
"default"
|
||||
],
|
||||
"default_subset": "default",
|
||||
"few_shot_num": 0,
|
||||
"few_shot_random": false,
|
||||
"train_split": "validation",
|
||||
"eval_split": "test",
|
||||
"prompt_template": "Answer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\nQuestion:\n{question}\nOptions:\n{choices}\n",
|
||||
"few_shot_prompt_template": "The following are multiple choice questions (with answers) about {subject}. Think step by step and then finish your answer with 'ANSWER: [LETTER]' (without quotes) where [LETTER] is the correct letter choice.\n\n{examples}\nAnswer the following multiple choice question. The last line of your response should be of the following format: 'ANSWER: [LETTER]' (without quotes) where [LETTER] is one of {letters}. Think step by step before answering.\n\nQuestion:\n{question}\nOptions:\n{choices}\n",
|
||||
"system_prompt": null,
|
||||
"query_template": null,
|
||||
"pretty_name": "MMLU-Pro (No Math)",
|
||||
"description": "MMLU-Pro benchmark excluding the math category.",
|
||||
"tags": [
|
||||
"MCQ",
|
||||
"Knowledge"
|
||||
],
|
||||
"filters": null,
|
||||
"metric_list": [
|
||||
"acc"
|
||||
],
|
||||
"aggregation": "mean",
|
||||
"shuffle": false,
|
||||
"shuffle_choices": false,
|
||||
"force_redownload": false,
|
||||
"review_timeout": null,
|
||||
"extra_params": {},
|
||||
"sandbox_config": {}
|
||||
}
|
||||
},
|
||||
"dataset_dir": "/fs/fast/u20240075/luoyashuo/modelscope_cache/datasets",
|
||||
"dataset_hub": "modelscope",
|
||||
"repeats": 1,
|
||||
"generation_config": {
|
||||
"batch_size": 2048,
|
||||
"max_tokens": 12000,
|
||||
"top_p": 1.0,
|
||||
"temperature": 1.0,
|
||||
"repetition_penalty": 1.0,
|
||||
"top_k": 50
|
||||
},
|
||||
"eval_type": "mock_llm",
|
||||
"eval_backend": "Native",
|
||||
"eval_config": null,
|
||||
"limit": null,
|
||||
"eval_batch_size": 2048,
|
||||
"use_cache": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052",
|
||||
"rerun_review": false,
|
||||
"work_dir": "/fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052",
|
||||
"no_timestamp": false,
|
||||
"ignore_errors": false,
|
||||
"debug": false,
|
||||
"seed": 42,
|
||||
"api_url": null,
|
||||
"timeout": null,
|
||||
"stream": null,
|
||||
"judge_strategy": "rule",
|
||||
"judge_worker_num": 1,
|
||||
"judge_model_args": {},
|
||||
"analysis_report": false,
|
||||
"use_sandbox": false,
|
||||
"sandbox_type": "docker",
|
||||
"sandbox_manager_config": {},
|
||||
"evalscope_version": "1.4.2"
|
||||
}
|
||||
2026-03-14 18:00:51 - evalscope - INFO: Start loading benchmark dataset: mmlu_pro_no_math
|
||||
2026-03-14 18:00:53 - evalscope - INFO: Start evaluating 1 subsets of the mmlu_pro_no_math: ['default']
|
||||
2026-03-14 18:00:53 - evalscope - INFO: Evaluating subset: default
|
||||
2026-03-14 18:00:53 - evalscope - INFO: Getting predictions for subset: default
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Reusing predictions from /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052/predictions/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math_default.jsonl, got 5028 predictions, remaining 5653 samples
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Processing 5653 samples, if data is large, it may take a while.
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Loading model for prediction...
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Model loaded successfully.
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Dispatcher: Worker-0 <- 3 prompts (pending=3, inflight=1)
|
||||
2026-03-14 18:00:54 - evalscope - INFO: Dispatcher: Worker-1 <- 3 prompts (pending=0, inflight=2)
|
||||
2026-03-14 18:01:56 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 0%| 0/5653 [Elapsed: 01:00 < Remaining: ?, ?it/s]
|
||||
2026-03-14 18:02:46 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1786, inflight=2)
|
||||
2026-03-14 18:02:57 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 0%| 3/5653 [Elapsed: 02:01 < Remaining: 174:51:19, 111.41s/it]
|
||||
2026-03-14 18:02:57 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1533, inflight=2)
|
||||
2026-03-14 18:03:58 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 0%| 6/5653 [Elapsed: 03:02 < Remaining: 37:47:14, 24.09s/it]
|
||||
2026-03-14 18:04:58 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 0%| 6/5653 [Elapsed: 04:03 < Remaining: 37:47:14, 24.09s/it]
|
||||
2026-03-14 18:05:59 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 0%| 6/5653 [Elapsed: 05:04 < Remaining: 37:47:14, 24.09s/it]
|
||||
2026-03-14 18:06:03 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:06:05 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:07:00 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 9%| 518/5653 [Elapsed: 06:04 < Remaining: 57:58, 1.48it/s]
|
||||
2026-03-14 18:08:01 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 9%| 518/5653 [Elapsed: 07:05 < Remaining: 57:58, 1.48it/s]
|
||||
2026-03-14 18:08:46 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:09:01 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 14%| 774/5653 [Elapsed: 08:05 < Remaining: 52:37, 1.54it/s]
|
||||
2026-03-14 18:09:36 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:10:02 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 18%| 1030/5653 [Elapsed: 09:06 < Remaining: 34:08, 2.26it/s]
|
||||
2026-03-14 18:11:02 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 18%| 1030/5653 [Elapsed: 10:07 < Remaining: 34:08, 2.26it/s]
|
||||
2026-03-14 18:11:19 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:11:58 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:12:03 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 27%| 1542/5653 [Elapsed: 11:07 < Remaining: 22:25, 3.06it/s]
|
||||
2026-03-14 18:13:03 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 27%| 1542/5653 [Elapsed: 12:07 < Remaining: 22:25, 3.06it/s]
|
||||
2026-03-14 18:14:04 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 27%| 1542/5653 [Elapsed: 13:08 < Remaining: 22:25, 3.06it/s]
|
||||
2026-03-14 18:14:35 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:15:04 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 32%| 1798/5653 [Elapsed: 14:09 < Remaining: 27:14, 2.36it/s]
|
||||
2026-03-14 18:15:55 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:16:05 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 36%| 2054/5653 [Elapsed: 15:10 < Remaining: 23:15, 2.58it/s]
|
||||
2026-03-14 18:17:06 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 36%| 2054/5653 [Elapsed: 16:10 < Remaining: 23:15, 2.58it/s]
|
||||
2026-03-14 18:18:06 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 36%| 2054/5653 [Elapsed: 17:11 < Remaining: 23:15, 2.58it/s]
|
||||
2026-03-14 18:19:07 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 36%| 2054/5653 [Elapsed: 18:11 < Remaining: 23:15, 2.58it/s]
|
||||
2026-03-14 18:19:08 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:20:07 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 41%| 2310/5653 [Elapsed: 19:12 < Remaining: 28:08, 1.98it/s]
|
||||
2026-03-14 18:21:08 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 41%| 2310/5653 [Elapsed: 20:12 < Remaining: 28:08, 1.98it/s]
|
||||
2026-03-14 18:22:08 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 41%| 2310/5653 [Elapsed: 21:13 < Remaining: 28:08, 1.98it/s]
|
||||
2026-03-14 18:22:31 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:23:09 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 22:13 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:24:09 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 23:14 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:25:10 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 24:14 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:26:10 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 25:14 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:27:10 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 26:15 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:28:11 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 27:15 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:29:11 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 28:15 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:30:11 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 29:16 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:31:12 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 30:16 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:32:12 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 45%| 2566/5653 [Elapsed: 31:16 < Remaining: 25:50, 1.99it/s]
|
||||
2026-03-14 18:32:40 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:33:13 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 50%| 2822/5653 [Elapsed: 32:17 < Remaining: 1:04:02, 1.36s/it]
|
||||
2026-03-14 18:34:13 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 50%| 2822/5653 [Elapsed: 33:17 < Remaining: 1:04:02, 1.36s/it]
|
||||
2026-03-14 18:35:13 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 50%| 2822/5653 [Elapsed: 34:18 < Remaining: 1:04:02, 1.36s/it]
|
||||
2026-03-14 18:35:37 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:36:14 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 35:18 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:37:14 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 36:19 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:38:15 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 37:19 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:39:15 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 38:19 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:40:15 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 39:20 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:41:16 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 40:20 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:42:16 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 41:20 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:43:16 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 42:20 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:44:16 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 43:21 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:45:17 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 44:21 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:46:17 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 54%| 3078/5653 [Elapsed: 45:21 < Remaining: 47:39, 1.11s/it]
|
||||
2026-03-14 18:46:22 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:47:18 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 59%| 3334/5653 [Elapsed: 46:22 < Remaining: 1:01:41, 1.60s/it]
|
||||
2026-03-14 18:48:18 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 59%| 3334/5653 [Elapsed: 47:22 < Remaining: 1:01:41, 1.60s/it]
|
||||
2026-03-14 18:48:38 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:49:19 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 48:23 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:50:19 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 49:23 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:51:19 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 50:23 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:52:19 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 51:24 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:53:20 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 52:24 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:54:20 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 53:24 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:55:20 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 64%| 3590/5653 [Elapsed: 54:24 < Remaining: 42:47, 1.24s/it]
|
||||
2026-03-14 18:55:22 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=1280, inflight=2)
|
||||
2026-03-14 18:56:21 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 68%| 3846/5653 [Elapsed: 55:25 < Remaining: 40:42, 1.35s/it]
|
||||
2026-03-14 18:56:26 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=1039, inflight=2)
|
||||
2026-03-14 18:57:21 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 73%| 4102/5653 [Elapsed: 56:26 < Remaining: 26:00, 1.01s/it]
|
||||
2026-03-14 18:58:21 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 73%| 4102/5653 [Elapsed: 57:26 < Remaining: 26:00, 1.01s/it]
|
||||
2026-03-14 18:59:22 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 73%| 4102/5653 [Elapsed: 58:26 < Remaining: 26:00, 1.01s/it]
|
||||
2026-03-14 18:59:37 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=783, inflight=2)
|
||||
2026-03-14 19:00:01 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=527, inflight=2)
|
||||
2026-03-14 19:00:22 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 59:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:01:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:00:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:02:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:01:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:03:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:02:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:04:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:03:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:05:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:04:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:06:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:05:27 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:07:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:06:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:08:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:07:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:09:23 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:08:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:10:24 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:09:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:11:24 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:10:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:12:24 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:11:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:13:24 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 82%| 4614/5653 [Elapsed: 1:12:28 < Remaining: 11:35, 1.49it/s]
|
||||
2026-03-14 19:14:16 - evalscope - INFO: Dispatcher: Worker-1 <- 256 prompts (pending=271, inflight=2)
|
||||
2026-03-14 19:14:24 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:13:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:15:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:14:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:16:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:15:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:17:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:16:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:18:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:17:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:19:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 86%| 4870/5653 [Elapsed: 1:18:29 < Remaining: 19:21, 1.48s/it]
|
||||
2026-03-14 19:20:19 - evalscope - INFO: Dispatcher: Worker-0 <- 256 prompts (pending=15, inflight=2)
|
||||
2026-03-14 19:20:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:19:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:21:25 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:20:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:22:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:21:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:23:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:22:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:24:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:23:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:25:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:24:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:26:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:25:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:27:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:26:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:28:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:27:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:29:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:28:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:30:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:29:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:31:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:30:30 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:32:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:31:31 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:33:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:32:31 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:34:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:33:31 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:35:26 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 91%| 5126/5653 [Elapsed: 1:34:31 < Remaining: 12:51, 1.46s/it]
|
||||
2026-03-14 19:35:36 - evalscope - INFO: Dispatcher: Worker-1 <- 15 prompts (pending=0, inflight=2)
|
||||
2026-03-14 19:36:27 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 95%| 5382/5653 [Elapsed: 1:35:31 < Remaining: 09:30, 2.10s/it]
|
||||
2026-03-14 19:37:27 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 95%| 5382/5653 [Elapsed: 1:36:31 < Remaining: 09:30, 2.10s/it]
|
||||
2026-03-14 19:38:27 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 95%| 5397/5653 [Elapsed: 1:37:31 < Remaining: 07:03, 1.66s/it]
|
||||
2026-03-14 19:39:27 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 95%| 5397/5653 [Elapsed: 1:38:31 < Remaining: 07:03, 1.66s/it]
|
||||
2026-03-14 19:40:27 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 95%| 5397/5653 [Elapsed: 1:39:31 < Remaining: 07:03, 1.66s/it]
|
||||
2026-03-14 19:41:09 - evalscope - INFO: Predicting[mmlu_pro_no_math@default]: 100%| 5653/5653 [Elapsed: 1:40:14 < Remaining: 00:00, 1.90s/it]
|
||||
2026-03-14 19:41:09 - evalscope - INFO: Finished getting predictions for subset: default.
|
||||
2026-03-14 19:41:09 - evalscope - INFO: Getting reviews for subset: default
|
||||
2026-03-14 19:41:09 - evalscope - INFO: Reviewing 10681 samples, if data is large, it may take a while.
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Reviewing[mmlu_pro_no_math@default]: 100%| 10681/10681 [Elapsed: 00:13 < Remaining: 00:00, 892.08it/s]
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Finished reviewing subset: default. Total reviewed: 10681
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Aggregating scores for subset: default
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Evaluating [mmlu_pro_no_math] 100%| 1/1 [Elapsed: 1:40:30 < Remaining: 00:00, 6030.39s/subset]
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Generating report...
|
||||
2026-03-14 19:41:23 - evalscope - INFO:
|
||||
mmlu_pro_no_math report table:
|
||||
+-----------------------------------------+------------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+==================+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | mmlu_pro_no_math | mean_acc | default | 10681 | 0.353 | default |
|
||||
+-----------------------------------------+------------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Skipping report analysis (`analysis_report=False`).
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Dump report to: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052/reports/llama3_3b_instruct_vallina_full_sft_30k/mmlu_pro_no_math.json
|
||||
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Benchmark mmlu_pro_no_math evaluation finished.
|
||||
2026-03-14 19:41:23 - evalscope - INFO: Overall report table:
|
||||
+-----------------------------------------+------------------+----------+----------+-------+---------+---------+
|
||||
| Model | Dataset | Metric | Subset | Num | Score | Cat.0 |
|
||||
+=========================================+==================+==========+==========+=======+=========+=========+
|
||||
| llama3_3b_instruct_vallina_full_sft_30k | mmlu_pro_no_math | mean_acc | default | 10681 | 0.353 | default |
|
||||
+-----------------------------------------+------------------+----------+----------+-------+---------+---------+
|
||||
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Finished evaluation for llama3_3b_instruct_vallina_full_sft_30k on ['mmlu_pro_no_math']
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Output directory: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath/20260314_121052
|
||||
2026-03-14 19:41:24 - evalscope - INFO: [进度条] MMLUProNoMath 评测完成 ✓
|
||||
2026-03-14 19:41:24 - evalscope - INFO: [断点续传] MMLUProNoMath 结果已保存: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_1/MMLUProNoMath_result.json
|
||||
2026-03-14 19:41:24 - evalscope - INFO: 完成评测 MMLUProNoMath (8/9)
|
||||
2026-03-14 19:41:24 - evalscope - INFO:
|
||||
==================================================
|
||||
2026-03-14 19:41:24 - evalscope - INFO: 正在评估 ARC (repeat: 1次) (剩余: 0个)
|
||||
2026-03-14 19:41:24 - evalscope - INFO: 模型: llama3_3b_instruct_vallina_full_sft_30k
|
||||
2026-03-14 19:41:24 - evalscope - INFO: 开始创建 benchmark ARC 的 TaskConfig
|
||||
2026-03-14 19:41:24 - evalscope - INFO: No model is provided, using DummyCustomModel for testing.
|
||||
2026-03-14 19:41:24 - evalscope - INFO: [ARC] worker_chunk_size=128
|
||||
2026-03-14 19:41:24 - evalscope - INFO: [ARC] TaskConfig创建完成: model=llama3_3b_instruct_vallina_full_sft_30k, datasets=['arc'], eval_batch_size=2048
|
||||
2026-03-14 19:41:24 - evalscope - INFO: 开始评测 ARC...
|
||||
2026-03-14 19:41:24 - evalscope - INFO: [进度条] ARC 开始评测
|
||||
2026-03-14 19:41:24 - evalscope - INFO: Args: Task config is provided with TaskConfig type.
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:fec245abdf31dfa69cd4fde0104cff1097fd0376b3f4814db198d16448f7ef7f
|
||||
size 284651943
|
||||
@@ -0,0 +1,34 @@
|
||||
{
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@mmlu_pro_no_math",
|
||||
"dataset_name": "mmlu_pro_no_math",
|
||||
"dataset_pretty_name": "MMLU-Pro (No Math)",
|
||||
"dataset_description": "MMLU-Pro benchmark excluding the math category.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.353,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.353,
|
||||
"num": 10681
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:bf112b871fb51d9d647e002ff1a20898cb618ebe00ff03779b0919acf9897e91
|
||||
size 152303121
|
||||
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"mmlu_pro_no_math": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@mmlu_pro_no_math",
|
||||
"dataset_name": "mmlu_pro_no_math",
|
||||
"dataset_pretty_name": "MMLU-Pro (No Math)",
|
||||
"dataset_description": "MMLU-Pro benchmark excluding the math category.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.353,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.353,
|
||||
"num": 10681
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
73
evalscope/version_20260313_110435/run_1/config.yaml
Normal file
73
evalscope/version_20260313_110435/run_1/config.yaml
Normal file
@@ -0,0 +1,73 @@
|
||||
API_KEY: sk-9ae88192c0e54080831b8a7408d2da8d
|
||||
BASE_URL: https://dashscope.aliyuncs.com/compatible-mode/v1
|
||||
benchmarks:
|
||||
AGIEvalMath:
|
||||
dataset_name: agieval_math
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
AIME24:
|
||||
dataset_name: aime24
|
||||
few_shot_num: 0
|
||||
max_tokens: 14000
|
||||
repeats: 8
|
||||
temperature: 0.6
|
||||
use_sandbox: false
|
||||
AIME25:
|
||||
dataset_name: aime25
|
||||
few_shot_num: 0
|
||||
max_tokens: 14000
|
||||
repeats: 8
|
||||
temperature: 0.6
|
||||
use_sandbox: false
|
||||
AMC:
|
||||
dataset_name: amc
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
subset: amc23
|
||||
temperature: 1.0
|
||||
ARC:
|
||||
dataset_name: arc
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
worker_chunk_size: 128
|
||||
GPQA:
|
||||
dataset_name: gpqa_extend
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
use_sandbox: false
|
||||
GSM8K:
|
||||
dataset_name: gsm8k
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
MATH500:
|
||||
dataset_name: math_500
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
MMLUProNoMath:
|
||||
dataset_name: mmlu_pro_no_math
|
||||
few_shot_num: 0
|
||||
max_tokens: 12000
|
||||
repeats: 1
|
||||
temperature: 1.0
|
||||
worker_chunk_size: 256
|
||||
model_path: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k
|
||||
sandbox:
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/tmp/evalscope_sandbox
|
||||
vllm:
|
||||
dp: 2
|
||||
enforce_eager: false
|
||||
gpu_memory_utilization: 0.9
|
||||
max_model_len: 16384
|
||||
tp: 1
|
||||
worker_chunk_size: 8
|
||||
366
evalscope/version_20260313_110435/run_1/detailed_results.json
Normal file
366
evalscope/version_20260313_110435/run_1/detailed_results.json
Normal file
@@ -0,0 +1,366 @@
|
||||
{
|
||||
"AIME25": {
|
||||
"aime25": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime25",
|
||||
"dataset_name": "aime25",
|
||||
"dataset_pretty_name": "AIME-2025",
|
||||
"dataset_description": "The AIME 2025 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "AIME2025-I",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
},
|
||||
{
|
||||
"name": "AIME2025-II",
|
||||
"score": 0.0167,
|
||||
"num": 120
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"AIME24": {
|
||||
"aime24": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@aime24",
|
||||
"dataset_name": "aime24",
|
||||
"dataset_pretty_name": "AIME-2024",
|
||||
"dataset_description": "The AIME 2024 benchmark is based on problems from the American Invitational Mathematics Examination, a prestigious high school mathematics competition. This benchmark tests a model's ability to solve challenging mathematics problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.0167,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 240,
|
||||
"score": 0.0167,
|
||||
"macro_score": 0.0167,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.0167,
|
||||
"num": 240
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"MATH500": {
|
||||
"math_500": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@math_500",
|
||||
"dataset_name": "math_500",
|
||||
"dataset_pretty_name": "MATH-500",
|
||||
"dataset_description": "MATH-500 is a benchmark for evaluating mathematical reasoning capabilities of AI models. It consists of 500 diverse math problems across five levels of difficulty, designed to test a model's ability to solve complex mathematical problems by generating step-by-step solutions and providing the correct final answer.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.464,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.464,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 500,
|
||||
"score": 0.464,
|
||||
"macro_score": 0.5229,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "Level 1",
|
||||
"score": 0.7442,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "Level 2",
|
||||
"score": 0.7222,
|
||||
"num": 90
|
||||
},
|
||||
{
|
||||
"name": "Level 3",
|
||||
"score": 0.5714,
|
||||
"num": 105
|
||||
},
|
||||
{
|
||||
"name": "Level 4",
|
||||
"score": 0.375,
|
||||
"num": 128
|
||||
},
|
||||
{
|
||||
"name": "Level 5",
|
||||
"score": 0.2015,
|
||||
"num": 134
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"AMC": {
|
||||
"amc": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@amc",
|
||||
"dataset_name": "amc",
|
||||
"dataset_pretty_name": "AMC",
|
||||
"dataset_description": "AMC (American Mathematics Competitions) is a series of mathematics competitions for high school students.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.1642,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1642,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 134,
|
||||
"score": 0.1642,
|
||||
"macro_score": 0.1629,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "amc22",
|
||||
"score": 0.1163,
|
||||
"num": 43
|
||||
},
|
||||
{
|
||||
"name": "amc23",
|
||||
"score": 0.2391,
|
||||
"num": 46
|
||||
},
|
||||
{
|
||||
"name": "amc24",
|
||||
"score": 0.1333,
|
||||
"num": 45
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"AGIEvalMath": {
|
||||
"agieval_math": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@agieval_math",
|
||||
"dataset_name": "agieval_math",
|
||||
"dataset_pretty_name": "AGIEval-Math",
|
||||
"dataset_description": "AGIEval-Math is a subset of AGIEval containing 1000 competition-level math problems drawn from the MATH dataset, covering algebra, geometry, number theory and more. Answers are clean numerical or symbolic expressions.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.472,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1000,
|
||||
"score": 0.472,
|
||||
"macro_score": 0.472,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.472,
|
||||
"num": 1000
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"GSM8K": {
|
||||
"gsm8k": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gsm8k",
|
||||
"dataset_name": "gsm8k",
|
||||
"dataset_pretty_name": "GSM8K",
|
||||
"dataset_description": "GSM8K (Grade School Math 8K) is a dataset of grade school math problems, designed to evaluate the mathematical reasoning abilities of AI models.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7521,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 1319,
|
||||
"score": 0.7521,
|
||||
"macro_score": 0.7521,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "main",
|
||||
"score": 0.7521,
|
||||
"num": 1319
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"GPQA": {
|
||||
"gpqa_extend": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@gpqa_extend",
|
||||
"dataset_name": "gpqa_extend",
|
||||
"dataset_pretty_name": "GPQA-Extended",
|
||||
"dataset_description": "GPQA Extended dataset for evaluating reasoning on graduate-level science problems.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.2454,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 546,
|
||||
"score": 0.2454,
|
||||
"macro_score": 0.2454,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "gpqa",
|
||||
"score": 0.2454,
|
||||
"num": 546
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"MMLUProNoMath": {
|
||||
"mmlu_pro_no_math": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@mmlu_pro_no_math",
|
||||
"dataset_name": "mmlu_pro_no_math",
|
||||
"dataset_pretty_name": "MMLU-Pro (No Math)",
|
||||
"dataset_description": "MMLU-Pro benchmark excluding the math category.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.353,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 10681,
|
||||
"score": 0.353,
|
||||
"macro_score": 0.353,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "default",
|
||||
"score": 0.353,
|
||||
"num": 10681
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
},
|
||||
"ARC": {
|
||||
"arc": {
|
||||
"name": "llama3_3b_instruct_vallina_full_sft_30k@arc",
|
||||
"dataset_name": "arc",
|
||||
"dataset_pretty_name": "ARC",
|
||||
"dataset_description": "The ARC (AI2 Reasoning Challenge) benchmark is designed to evaluate the reasoning capabilities of AI models through multiple-choice questions derived from science exams. It includes two subsets: ARC-Easy and ARC-Challenge, which vary in difficulty.",
|
||||
"model_name": "llama3_3b_instruct_vallina_full_sft_30k",
|
||||
"score": 0.7903,
|
||||
"metrics": [
|
||||
{
|
||||
"name": "mean_acc",
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7903,
|
||||
"categories": [
|
||||
{
|
||||
"name": [
|
||||
"default"
|
||||
],
|
||||
"num": 3548,
|
||||
"score": 0.7903,
|
||||
"macro_score": 0.7712,
|
||||
"subsets": [
|
||||
{
|
||||
"name": "ARC-Easy",
|
||||
"score": 0.8274,
|
||||
"num": 2376
|
||||
},
|
||||
{
|
||||
"name": "ARC-Challenge",
|
||||
"score": 0.715,
|
||||
"num": 1172
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"analysis": "N/A"
|
||||
}
|
||||
}
|
||||
}
|
||||
BIN
evalscope/version_20260313_110435/run_1/summary.xlsx
Normal file
BIN
evalscope/version_20260313_110435/run_1/summary.xlsx
Normal file
Binary file not shown.
@@ -0,0 +1,78 @@
|
||||
analysis_report: false
|
||||
api_url: null
|
||||
chat_template: null
|
||||
dataset_args:
|
||||
agieval_math:
|
||||
aggregation: mean
|
||||
dataset_id: hails/agieval-math
|
||||
default_subset: default
|
||||
description: AGIEval-Math is a subset of AGIEval containing 1000 competition-level
|
||||
math problems drawn from the MATH dataset, covering algebra, geometry, number
|
||||
theory and more. Answers are clean numerical or symbolic expressions.
|
||||
eval_split: test
|
||||
extra_params: {}
|
||||
few_shot_num: 0
|
||||
few_shot_prompt_template: null
|
||||
few_shot_random: false
|
||||
filters: null
|
||||
force_redownload: false
|
||||
metric_list:
|
||||
- acc:
|
||||
numeric: true
|
||||
name: agieval_math
|
||||
output_types:
|
||||
- generation
|
||||
pretty_name: AGIEval-Math
|
||||
prompt_template: '{question}
|
||||
|
||||
Please reason step by step, and put your final answer within \boxed{{}}.'
|
||||
query_template: null
|
||||
review_timeout: null
|
||||
sandbox_config: {}
|
||||
shuffle: false
|
||||
shuffle_choices: false
|
||||
subset_list:
|
||||
- default
|
||||
system_prompt: Please reason step by step to solve the problem. Put your final
|
||||
answer in \boxed{}.
|
||||
tags:
|
||||
- Math
|
||||
- Reasoning
|
||||
train_split: null
|
||||
dataset_dir: /fs/fast/u20240075/luoyashuo/modelscope_cache/datasets
|
||||
dataset_hub: modelscope
|
||||
datasets:
|
||||
- agieval_math
|
||||
debug: false
|
||||
eval_backend: Native
|
||||
eval_batch_size: 2048
|
||||
eval_config: null
|
||||
eval_type: mock_llm
|
||||
evalscope_version: 1.4.2
|
||||
generation_config:
|
||||
batch_size: 2048
|
||||
max_tokens: 12000
|
||||
repetition_penalty: 1.0
|
||||
temperature: 1.0
|
||||
top_k: 50
|
||||
top_p: 1.0
|
||||
ignore_errors: false
|
||||
judge_model_args: {}
|
||||
judge_strategy: rule
|
||||
judge_worker_num: 1
|
||||
limit: null
|
||||
model: VLLMOfflineModelAPI
|
||||
model_args: {}
|
||||
model_id: llama3_3b_instruct_vallina_full_sft_30k
|
||||
model_task: text_generation
|
||||
no_timestamp: false
|
||||
repeats: 1
|
||||
rerun_review: false
|
||||
sandbox_manager_config: {}
|
||||
sandbox_type: docker
|
||||
seed: 42
|
||||
stream: null
|
||||
timeout: null
|
||||
use_cache: null
|
||||
use_sandbox: false
|
||||
work_dir: /fs/fast/u20240075/luoyashuo/cdad/checkpoints/vallina_sft/llama3_3b_instruct_vallina_full_sft_30k/evalscope/version_20260313_110435/run_2/AGIEvalMath/20260314_222836
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user