commit de7018a39754c970cc42d6267100014c6cb61009 Author: ModelHub XC Date: Sat Sep 5 18:51:17 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: agarwalanu3103/clarify-rl-grpo-qwen3-1-7b-run7 Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..247726c --- /dev/null +++ b/README.md @@ -0,0 +1,70 @@ +--- +base_model: Qwen/Qwen3-1.7B +library_name: transformers +model_name: clarify-rl-grpo-qwen3-1-7b-run7 +tags: +- generated_from_trainer +- grpo +- trl +- hf_jobs +- trackio +- trackio:https://huggingface.co/spaces/agarwalanu3103/huggingface-static-423401 +licence: license +--- + +# Model Card for clarify-rl-grpo-qwen3-1-7b-run7 + +This model is a fine-tuned version of [Qwen/Qwen3-1.7B](https://huggingface.co/Qwen/Qwen3-1.7B). +It has been trained using [TRL](https://github.com/huggingface/trl). + +## Quick start + +```python +from transformers import pipeline + +question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?" +generator = pipeline("text-generation", model="agarwalanu3103/clarify-rl-grpo-qwen3-1-7b-run7", device="cuda") +output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0] +print(output["generated_text"]) +``` + +## Training procedure + + + + + +This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300). + +### Framework versions + +- TRL: 1.2.0 +- Transformers: 5.7.0.dev0 +- Pytorch: 2.8.0 +- Datasets: 4.8.4 +- Tokenizers: 0.22.2 + +## Citations + +Cite GRPO as: + +```bibtex +@article{shao2024deepseekmath, + title = {{DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models}}, + author = {Zhihong Shao and Peiyi Wang and Qihao Zhu and Runxin Xu and Junxiao Song and Mingchuan Zhang and Y. K. Li and Y. Wu and Daya Guo}, + year = 2024, + eprint = {arXiv:2402.03300}, +} +``` + +Cite TRL as: + +```bibtex +@software{vonwerra2020trl, + title = {{TRL: Transformers Reinforcement Learning}}, + author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin}, + license = {Apache-2.0}, + url = {https://github.com/huggingface/trl}, + year = {2020} +} +``` \ No newline at end of file diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000..01be9b3 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,89 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0].role == 'system' %} + {{- messages[0].content + '\n\n' }} + {%- endif %} + {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0].role == 'system' %} + {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('') and message.content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} +{%- endfor %} +{%- for message in messages %} + {%- if message.content is string %} + {%- set content = message.content %} + {%- else %} + {%- set content = '' %} + {%- endif %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- if loop.index0 > ns.last_query_index %} + {%- if loop.last or (not loop.last and reasoning_content) %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content.strip('\n') + '\n\n\n' + content.lstrip('\n') }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls %} + {%- for tool_call in message.tool_calls %} + {%- if (loop.first and content) or (not loop.first) %} + {{- '\n' }} + {%- endif %} + {%- if tool_call.function %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {%- if tool_call.arguments is string %} + {{- tool_call.arguments }} + {%- else %} + {{- tool_call.arguments | tojson }} + {%- endif %} + {{- '}\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/completions/completions_00001.parquet b/completions/completions_00001.parquet new file mode 100644 index 0000000..79d1d5f --- /dev/null +++ b/completions/completions_00001.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:34a9c5094876107a6ae3440b941804a984c4a773b023a16f0fe39583142dcdd4 +size 24549 diff --git a/completions/completions_00002.parquet b/completions/completions_00002.parquet new file mode 100644 index 0000000..f4f6e36 --- /dev/null +++ b/completions/completions_00002.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a4a842fb2b939dbce812bcaca6e7de45e55bb6e02a039b2a35a0777c66fd7273 +size 22871 diff --git a/completions/completions_00003.parquet b/completions/completions_00003.parquet new file mode 100644 index 0000000..f8b20e7 --- /dev/null +++ b/completions/completions_00003.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:130682cbffca0f71307b0bb1f7eefc570136afb4604d4359ef20dce29ed1b931 +size 26850 diff --git a/completions/completions_00004.parquet b/completions/completions_00004.parquet new file mode 100644 index 0000000..31fb9e2 --- /dev/null +++ b/completions/completions_00004.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:60d1bf6734a116f814cd4c4147e87e2c373c380e40eeca002a12d134df2dec05 +size 27251 diff --git a/completions/completions_00005.parquet b/completions/completions_00005.parquet new file mode 100644 index 0000000..1725c03 --- /dev/null +++ b/completions/completions_00005.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1704905d6fcf156fbd34d3b0a4aad67250c784cd0f96763678d387da0909e1f +size 22993 diff --git a/completions/completions_00006.parquet b/completions/completions_00006.parquet new file mode 100644 index 0000000..78fbf55 --- /dev/null +++ b/completions/completions_00006.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dedd4a7ab500b774dd57aaea1b25b545c15e6de22ecbbb41bf059792195fbbfd +size 25178 diff --git a/completions/completions_00007.parquet b/completions/completions_00007.parquet new file mode 100644 index 0000000..7741d2f --- /dev/null +++ b/completions/completions_00007.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8a162a1630eafb2babaac03f306f2fd2db7e2dbff0e325aba19596560c9a9636 +size 23248 diff --git a/completions/completions_00008.parquet b/completions/completions_00008.parquet new file mode 100644 index 0000000..3427c20 --- /dev/null +++ b/completions/completions_00008.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c402b7d5503ee0c0b249309bdc5949c69978d1f96d16994bcd688fc975492a1 +size 22590 diff --git a/completions/completions_00009.parquet b/completions/completions_00009.parquet new file mode 100644 index 0000000..1221d20 --- /dev/null +++ b/completions/completions_00009.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad3f06634c107be0bfd66acd40445482cb35068abe50b558b350b771e41be7d3 +size 28004 diff --git a/completions/completions_00010.parquet b/completions/completions_00010.parquet new file mode 100644 index 0000000..26dee2f --- /dev/null +++ b/completions/completions_00010.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d6cc46ab785446484d7d8e6925d5ad33995950657e78b02e051b906f33d67fd +size 22897 diff --git a/completions/completions_00011.parquet b/completions/completions_00011.parquet new file mode 100644 index 0000000..b6f07d4 --- /dev/null +++ b/completions/completions_00011.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8993df5c4ebd0e08e9f80ee6e99e13a2f1fdf551e793dbe81d715c41bc41fe69 +size 23006 diff --git a/completions/completions_00012.parquet b/completions/completions_00012.parquet new file mode 100644 index 0000000..1cb63e4 --- /dev/null +++ b/completions/completions_00012.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3680bf3c6af643e9a38f0dd4b364f5d87f0c3cb5f281d2e144d180190b7312ca +size 25169 diff --git a/completions/completions_00013.parquet b/completions/completions_00013.parquet new file mode 100644 index 0000000..64e88c5 --- /dev/null +++ b/completions/completions_00013.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6de659c83f70d2a63584c2e72a9666f9131d52440b28912d1f2d79d6b978d4bf +size 22899 diff --git a/completions/completions_00014.parquet b/completions/completions_00014.parquet new file mode 100644 index 0000000..07e7c94 --- /dev/null +++ b/completions/completions_00014.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:73365eeddbc8267ae67e41f5c16410c2df58420b147586a21e659ef4c176af8c +size 28860 diff --git a/completions/completions_00015.parquet b/completions/completions_00015.parquet new file mode 100644 index 0000000..c6233c6 --- /dev/null +++ b/completions/completions_00015.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2e7e57bd8efe7ea6b9c19c272bcabd1aabe91a98330ed52eba733cfb3fcd85fa +size 26890 diff --git a/completions/completions_00016.parquet b/completions/completions_00016.parquet new file mode 100644 index 0000000..9f3a612 --- /dev/null +++ b/completions/completions_00016.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:61eb3d1ff8d047c080e96117385dd5e7b5aa3b87a3961d89f78b24ffc41a1b82 +size 23076 diff --git a/completions/completions_00017.parquet b/completions/completions_00017.parquet new file mode 100644 index 0000000..c8e60be --- /dev/null +++ b/completions/completions_00017.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:97a68a140e6d75c7a601fc059bd236e5ee3b67639dc0ad4ff27100e4fc3a427b +size 27373 diff --git a/completions/completions_00018.parquet b/completions/completions_00018.parquet new file mode 100644 index 0000000..651511d --- /dev/null +++ b/completions/completions_00018.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:88c5392030d879c5066f6bf54f6e804a4829152870604ddd39ce01bec1efe812 +size 23163 diff --git a/completions/completions_00019.parquet b/completions/completions_00019.parquet new file mode 100644 index 0000000..dc0e72a --- /dev/null +++ b/completions/completions_00019.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e550753a18ef3fbe3b97b470285d3dcdb3e4ad9f7d233ac6d4faf488956e2eb +size 32582 diff --git a/completions/completions_00020.parquet b/completions/completions_00020.parquet new file mode 100644 index 0000000..56f8629 --- /dev/null +++ b/completions/completions_00020.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0754e14b8b31020f9564db87ebcc2ecb8374aaf2e50c927fe0ba3370b9e685b4 +size 22281 diff --git a/completions/completions_00021.parquet b/completions/completions_00021.parquet new file mode 100644 index 0000000..65af396 --- /dev/null +++ b/completions/completions_00021.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1b5db88971c5b09b4aff85be23bd543939070aac392b9b035724f607f7915a8 +size 23124 diff --git a/completions/completions_00022.parquet b/completions/completions_00022.parquet new file mode 100644 index 0000000..3067606 --- /dev/null +++ b/completions/completions_00022.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9bcbf455377d2fff368041d1d4d64b04d1bc7a8b4fb7c873318f6e3164c8b2bc +size 28459 diff --git a/completions/completions_00023.parquet b/completions/completions_00023.parquet new file mode 100644 index 0000000..6a6e643 --- /dev/null +++ b/completions/completions_00023.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:acfe392ea04c2fe66b28af432ee4ff616e7b708cbe5c73c0530f87ee42e8660e +size 23166 diff --git a/completions/completions_00024.parquet b/completions/completions_00024.parquet new file mode 100644 index 0000000..981f04e --- /dev/null +++ b/completions/completions_00024.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:819e8d82e08ca66e5052ccccc316678a7737dd348b56019e1c733849235bace7 +size 23416 diff --git a/completions/completions_00025.parquet b/completions/completions_00025.parquet new file mode 100644 index 0000000..eb3ebe2 --- /dev/null +++ b/completions/completions_00025.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04c6fc095892520eb3ad11b12e481cdc3a33dd86fbe8a5ba572f07f8bc4a85e6 +size 23159 diff --git a/completions/completions_00026.parquet b/completions/completions_00026.parquet new file mode 100644 index 0000000..edcb960 --- /dev/null +++ b/completions/completions_00026.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:140f48c1ac2af6e7b1136aedc190ae6fc9c7288d46160cc81207e504ef1b154a +size 22880 diff --git a/completions/completions_00027.parquet b/completions/completions_00027.parquet new file mode 100644 index 0000000..8278915 --- /dev/null +++ b/completions/completions_00027.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:51fc2e5f0af74225d7568721910a3b41a647c050156878eeca5dcc48100b2cca +size 33354 diff --git a/completions/completions_00028.parquet b/completions/completions_00028.parquet new file mode 100644 index 0000000..f6c2109 --- /dev/null +++ b/completions/completions_00028.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8af74e590368d1ea7fa545153651c11131d4e8002c4fffffa0ae163e3de68bc3 +size 28320 diff --git a/completions/completions_00029.parquet b/completions/completions_00029.parquet new file mode 100644 index 0000000..a010fe0 --- /dev/null +++ b/completions/completions_00029.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4389e08d03be11522d57c5025aa63e1499fd99fbe1ad9adf5ebbeb5a6b4e141c +size 26306 diff --git a/completions/completions_00030.parquet b/completions/completions_00030.parquet new file mode 100644 index 0000000..c5155ae --- /dev/null +++ b/completions/completions_00030.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:85a5f3b35edce66fe5bc77490d8bf6007ea73e7b1c8221c305807e2f01d7b792 +size 28348 diff --git a/completions/completions_00031.parquet b/completions/completions_00031.parquet new file mode 100644 index 0000000..82d5ae7 --- /dev/null +++ b/completions/completions_00031.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b625ebbb26cd15d0d8e6a031094fb3a0b81c7556645082695a5965ce212a5526 +size 26646 diff --git a/completions/completions_00032.parquet b/completions/completions_00032.parquet new file mode 100644 index 0000000..ff33828 --- /dev/null +++ b/completions/completions_00032.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4bb5b5f756d0cdc0a5dbef65443adce40622fc8e635e6606c9e1a2313b8e191d +size 23032 diff --git a/completions/completions_00033.parquet b/completions/completions_00033.parquet new file mode 100644 index 0000000..9ef1492 --- /dev/null +++ b/completions/completions_00033.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8db87708c49771c26a54c8a6ffcadcf92552b518a440210d1fa1be85ffcdbf4b +size 22944 diff --git a/completions/completions_00034.parquet b/completions/completions_00034.parquet new file mode 100644 index 0000000..cfa761c --- /dev/null +++ b/completions/completions_00034.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cd508897f05c1d365751c6bdca8cefdc42b94980d693a59314a35a8d39db30e7 +size 22806 diff --git a/completions/completions_00035.parquet b/completions/completions_00035.parquet new file mode 100644 index 0000000..ac6937e --- /dev/null +++ b/completions/completions_00035.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4be7f186f5ef1e76ab8b92738a270cac737b0dd5e5b2fd06d778207d533b7bd3 +size 22870 diff --git a/completions/completions_00036.parquet b/completions/completions_00036.parquet new file mode 100644 index 0000000..d176e77 --- /dev/null +++ b/completions/completions_00036.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f215b38cb8f3e8ae8386fbf71f502b68e3653ed6051db3da1386d5938dc24cc0 +size 22854 diff --git a/completions/completions_00037.parquet b/completions/completions_00037.parquet new file mode 100644 index 0000000..3219bd8 --- /dev/null +++ b/completions/completions_00037.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93d7b4a2fc1bf12163c09f72e3902c29273b1023a5d734c13f77fb74a9c94140 +size 28933 diff --git a/completions/completions_00038.parquet b/completions/completions_00038.parquet new file mode 100644 index 0000000..39bf754 --- /dev/null +++ b/completions/completions_00038.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:439b93c02c3dc4955cfe5161e3bd47f032dccbf288415fd07b680761ac9cac71 +size 28031 diff --git a/completions/completions_00039.parquet b/completions/completions_00039.parquet new file mode 100644 index 0000000..1491396 --- /dev/null +++ b/completions/completions_00039.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4a0eb6dc392391e475b0bb4eb53d9f8d186092947e9e9ef1835a8719993b3aaf +size 27840 diff --git a/completions/completions_00040.parquet b/completions/completions_00040.parquet new file mode 100644 index 0000000..ed5d8b1 --- /dev/null +++ b/completions/completions_00040.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:238a7ea807c95783f08a2cf7d6050338ca959933259b9ab6c992d7400af4751a +size 22723 diff --git a/completions/completions_00041.parquet b/completions/completions_00041.parquet new file mode 100644 index 0000000..c952973 --- /dev/null +++ b/completions/completions_00041.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:35654974fed54693db5f2c4932cdce2e8f3054451cbaa452d5a28142a34df0cc +size 28568 diff --git a/completions/completions_00042.parquet b/completions/completions_00042.parquet new file mode 100644 index 0000000..9a2f127 --- /dev/null +++ b/completions/completions_00042.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5e96dbadc2734f237544bee8e109853ca047650b4b126f49aac183dec1482628 +size 31854 diff --git a/completions/completions_00043.parquet b/completions/completions_00043.parquet new file mode 100644 index 0000000..cf232df --- /dev/null +++ b/completions/completions_00043.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fdf1786567b14d5df7df24d89df6c3f8106feabe095bbed08bacaf64974221f2 +size 29634 diff --git a/completions/completions_00044.parquet b/completions/completions_00044.parquet new file mode 100644 index 0000000..2f47650 --- /dev/null +++ b/completions/completions_00044.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e95c8e0975ab495525de9d8790c9a95aaa5a54726f694189465a23498b1f7f38 +size 28051 diff --git a/completions/completions_00045.parquet b/completions/completions_00045.parquet new file mode 100644 index 0000000..4264e1a --- /dev/null +++ b/completions/completions_00045.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e314c78f1ebd421df9ddb170440425fd469439de27ffe83d6aaa4add679bb9b3 +size 28301 diff --git a/completions/completions_00046.parquet b/completions/completions_00046.parquet new file mode 100644 index 0000000..f063871 --- /dev/null +++ b/completions/completions_00046.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:482faa263ed43ed9df93be0b7d1c214adfca0e0228ddb323c4bbab8c358811a2 +size 22748 diff --git a/completions/completions_00047.parquet b/completions/completions_00047.parquet new file mode 100644 index 0000000..50a98f9 --- /dev/null +++ b/completions/completions_00047.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:57093db41637d60dec953fcd2c1b782e3bdf50090b57b106b5fa322e0a3fda72 +size 22777 diff --git a/completions/completions_00048.parquet b/completions/completions_00048.parquet new file mode 100644 index 0000000..526dc3f --- /dev/null +++ b/completions/completions_00048.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b2112a497b3abae770b64bc9626eeb38b9b84712cb5ad146323a4cdc86dceb78 +size 32694 diff --git a/completions/completions_00049.parquet b/completions/completions_00049.parquet new file mode 100644 index 0000000..20a152e --- /dev/null +++ b/completions/completions_00049.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27e1d927e26fdca4fe480ab1e85300ba2396164b5c2c02ac0385d951f8c9337a +size 31440 diff --git a/completions/completions_00050.parquet b/completions/completions_00050.parquet new file mode 100644 index 0000000..3dc8c59 --- /dev/null +++ b/completions/completions_00050.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:78e4504f1ef3528b2a7624d7a1d1490a3f80fc5058510b193036b8fe5b25ce44 +size 27437 diff --git a/completions/completions_00051.parquet b/completions/completions_00051.parquet new file mode 100644 index 0000000..71a6ca3 --- /dev/null +++ b/completions/completions_00051.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:545aed02ffa551d0fe0e1febaa49578ea22fd878b22f0a523b918c9503f178f4 +size 26466 diff --git a/completions/completions_00052.parquet b/completions/completions_00052.parquet new file mode 100644 index 0000000..90e3962 --- /dev/null +++ b/completions/completions_00052.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f4620ae456212d47f81daf961f449d94ed72f94aff2d7c852db815aa5821b6dc +size 31186 diff --git a/completions/completions_00053.parquet b/completions/completions_00053.parquet new file mode 100644 index 0000000..4946c69 --- /dev/null +++ b/completions/completions_00053.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb29106c1f38ea298de6de4d94f75094a499d65bf7698935143fd948704b1a09 +size 30679 diff --git a/completions/completions_00054.parquet b/completions/completions_00054.parquet new file mode 100644 index 0000000..c83c7fd --- /dev/null +++ b/completions/completions_00054.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dcd9fc0fd20e2e0571fd00048dff7d87a6e198e43c80ca6f04cf9923d0b333c4 +size 22959 diff --git a/completions/completions_00055.parquet b/completions/completions_00055.parquet new file mode 100644 index 0000000..2181887 --- /dev/null +++ b/completions/completions_00055.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:847e0c17b40fbdedeb19c0aab2f52f5ba8371ca95041ed0b086c1bc1ed5e2aed +size 27757 diff --git a/completions/completions_00056.parquet b/completions/completions_00056.parquet new file mode 100644 index 0000000..80e8758 --- /dev/null +++ b/completions/completions_00056.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:31005b88f04b465948188694bc12c2db8fb12faab9999c0c3dfe77d735b96241 +size 34002 diff --git a/completions/completions_00057.parquet b/completions/completions_00057.parquet new file mode 100644 index 0000000..8420372 --- /dev/null +++ b/completions/completions_00057.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc0e1c4b5a29d03f827ce14dfac418b5042650943762707ced596e3e5cfab1cf +size 28288 diff --git a/completions/completions_00058.parquet b/completions/completions_00058.parquet new file mode 100644 index 0000000..02fa91c --- /dev/null +++ b/completions/completions_00058.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dfabdb84c6c2fe9b6c344b05a09c1c072c2c8e44ab71b3343f92857dcbae34b1 +size 27432 diff --git a/completions/completions_00059.parquet b/completions/completions_00059.parquet new file mode 100644 index 0000000..f9525a3 --- /dev/null +++ b/completions/completions_00059.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a4c93ee152f2dd526e24a6aa9c13f27f6d53ab34a4b5e4a70f8d9306b726fc9 +size 27992 diff --git a/completions/completions_00060.parquet b/completions/completions_00060.parquet new file mode 100644 index 0000000..ed49aba --- /dev/null +++ b/completions/completions_00060.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:70f265a60203c89543bb3113baa721ccf15aa1222a9686e48b2c07d33e178d0e +size 28015 diff --git a/completions/completions_00061.parquet b/completions/completions_00061.parquet new file mode 100644 index 0000000..10b5f44 --- /dev/null +++ b/completions/completions_00061.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac5857d9cb8386f6420974eb8f4ab698b0508eb34a59be61f5d77707f0cf1da7 +size 28363 diff --git a/completions/completions_00062.parquet b/completions/completions_00062.parquet new file mode 100644 index 0000000..e6c5fca --- /dev/null +++ b/completions/completions_00062.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:295949ee5ef549c8b316cd444a05d2f3f3e8f288ffc3a976786bb0da64d56825 +size 23104 diff --git a/completions/completions_00063.parquet b/completions/completions_00063.parquet new file mode 100644 index 0000000..03791fd --- /dev/null +++ b/completions/completions_00063.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:035282344682d66d86ae2988cf908fae03fa12ca9858696ef3cf101d2aaae71a +size 33868 diff --git a/completions/completions_00064.parquet b/completions/completions_00064.parquet new file mode 100644 index 0000000..05a5a5d --- /dev/null +++ b/completions/completions_00064.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:683d2d711acf906a3e794ca207759ff0b77b92561c0d702dc37d3aae687ec0cb +size 22871 diff --git a/completions/completions_00065.parquet b/completions/completions_00065.parquet new file mode 100644 index 0000000..2d3208e --- /dev/null +++ b/completions/completions_00065.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a388495c07cf8102d07c62127a9dfed744e2701e0a16426ef432ebc8d5280a32 +size 31195 diff --git a/completions/completions_00066.parquet b/completions/completions_00066.parquet new file mode 100644 index 0000000..63a98c7 --- /dev/null +++ b/completions/completions_00066.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de06ac1fc4d8d5b60d22ddf306e242494406161db25b202699f23d75b5c14390 +size 27421 diff --git a/completions/completions_00067.parquet b/completions/completions_00067.parquet new file mode 100644 index 0000000..36ce0ba --- /dev/null +++ b/completions/completions_00067.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4f329f62387da366babbf6e0e9bba2b4db0c0afaa371a5fb748fbb44554bac9e +size 27289 diff --git a/completions/completions_00068.parquet b/completions/completions_00068.parquet new file mode 100644 index 0000000..8d6546c --- /dev/null +++ b/completions/completions_00068.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c49e370876cada9f86dc9f71783c416268c284855f215298129681a15ceeffe8 +size 26098 diff --git a/completions/completions_00069.parquet b/completions/completions_00069.parquet new file mode 100644 index 0000000..92c64f8 --- /dev/null +++ b/completions/completions_00069.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eb7562e76c7f7d4e5405e77b76597a8c1c1b606387cc02ed88355d60d6d453f7 +size 30534 diff --git a/completions/completions_00070.parquet b/completions/completions_00070.parquet new file mode 100644 index 0000000..5edfcd8 --- /dev/null +++ b/completions/completions_00070.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6a4f3a1972f672472299a6b0098c426cff645af3c3b8b744cb074f9f10910ec3 +size 22711 diff --git a/completions/completions_00071.parquet b/completions/completions_00071.parquet new file mode 100644 index 0000000..2faa316 --- /dev/null +++ b/completions/completions_00071.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b504c62dd6efc467629e92222d608f9b7008667a378f0113bfcc083af4ffad0 +size 22656 diff --git a/completions/completions_00072.parquet b/completions/completions_00072.parquet new file mode 100644 index 0000000..b41e090 --- /dev/null +++ b/completions/completions_00072.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a37167114b742c040bdc4112125c808e1cf0e9aca697e3f96933250c1ce625ad +size 28933 diff --git a/completions/completions_00073.parquet b/completions/completions_00073.parquet new file mode 100644 index 0000000..5d220cb --- /dev/null +++ b/completions/completions_00073.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4d1e2f0104df5d8a2b252630619da0a711e4998c2044acec41daef281bbf6c8a +size 30853 diff --git a/completions/completions_00074.parquet b/completions/completions_00074.parquet new file mode 100644 index 0000000..4ce51c9 --- /dev/null +++ b/completions/completions_00074.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4fd211a7fe4129c0bcd7a2e75cd55fc03d70e07220e26f4472909fcb19b7aff0 +size 29478 diff --git a/completions/completions_00075.parquet b/completions/completions_00075.parquet new file mode 100644 index 0000000..cf51c20 --- /dev/null +++ b/completions/completions_00075.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8988ad64a8ea5c459634b8ac9ff5eb16321871a6422fd1df92b35afb2eb254a9 +size 27975 diff --git a/completions/completions_00076.parquet b/completions/completions_00076.parquet new file mode 100644 index 0000000..eb0da59 --- /dev/null +++ b/completions/completions_00076.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e9704961de8d28c258d3adc74a980eb83b2057f2de0e69b5b933a53cb3e0495d +size 31643 diff --git a/completions/completions_00077.parquet b/completions/completions_00077.parquet new file mode 100644 index 0000000..ae1ef81 --- /dev/null +++ b/completions/completions_00077.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8a62d89c6819b4ed5aa77ec012e1be9870639e743dc23071ca878fc37e4e7c1c +size 27543 diff --git a/completions/completions_00078.parquet b/completions/completions_00078.parquet new file mode 100644 index 0000000..d228f1f --- /dev/null +++ b/completions/completions_00078.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f297938376a294d9bc9aa5f5d8d6bc017b2f39e658970fe419e839334c0982c1 +size 33885 diff --git a/completions/completions_00079.parquet b/completions/completions_00079.parquet new file mode 100644 index 0000000..0aeb082 --- /dev/null +++ b/completions/completions_00079.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1cfebcff92e682a0f791cb7e6bfc4634d018d696c5e7ef908c1477e7a46cc5f +size 31180 diff --git a/completions/completions_00080.parquet b/completions/completions_00080.parquet new file mode 100644 index 0000000..1ee9667 --- /dev/null +++ b/completions/completions_00080.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3023b6fbcd12bba6bde3abec6a9a501d3dfd0ca8945a8dfef72bd16138e090e3 +size 31936 diff --git a/completions/completions_00081.parquet b/completions/completions_00081.parquet new file mode 100644 index 0000000..5610168 --- /dev/null +++ b/completions/completions_00081.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8903dece9328ef09700f746b79ec429ba24310daee3516b7b238b6fa4deeb13 +size 32763 diff --git a/completions/completions_00082.parquet b/completions/completions_00082.parquet new file mode 100644 index 0000000..0975ef1 --- /dev/null +++ b/completions/completions_00082.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a620082439908705b710a24a3a70cbaf8953552dcb09d641f358e5fd34095984 +size 23142 diff --git a/completions/completions_00083.parquet b/completions/completions_00083.parquet new file mode 100644 index 0000000..f0a3e1f --- /dev/null +++ b/completions/completions_00083.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02801b9f215e360e14fabc6e465b47372b61b89dbcffee82d80bc5d814e2edb4 +size 28142 diff --git a/completions/completions_00084.parquet b/completions/completions_00084.parquet new file mode 100644 index 0000000..9e03422 --- /dev/null +++ b/completions/completions_00084.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1684ca86a8d743c5e289d3c27765d0b408dd2125a279f9dee669efa6de0e401f +size 30358 diff --git a/completions/completions_00085.parquet b/completions/completions_00085.parquet new file mode 100644 index 0000000..6bac339 --- /dev/null +++ b/completions/completions_00085.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3eea29a1e5021bf9940f5f7ffcc68fc6a309bb9204d1dd0c67b229ddead2df50 +size 28288 diff --git a/completions/completions_00086.parquet b/completions/completions_00086.parquet new file mode 100644 index 0000000..e0bca8b --- /dev/null +++ b/completions/completions_00086.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3d6cce04ddef15dc673f61420b3e2aa0de0b4b022665c43b9734a69a19d1800 +size 23054 diff --git a/completions/completions_00087.parquet b/completions/completions_00087.parquet new file mode 100644 index 0000000..c9f8b03 --- /dev/null +++ b/completions/completions_00087.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b91f3dd0ca0c05f92895fcbe5e72a8206bb7057d9289480a12dc243c43865bb4 +size 26948 diff --git a/completions/completions_00088.parquet b/completions/completions_00088.parquet new file mode 100644 index 0000000..64e94c3 --- /dev/null +++ b/completions/completions_00088.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7a77e25bb08a4f41d36a836173e940a5b68927859ae09667d4cbccd3dbabe67 +size 27133 diff --git a/completions/completions_00089.parquet b/completions/completions_00089.parquet new file mode 100644 index 0000000..7c66dc3 --- /dev/null +++ b/completions/completions_00089.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1be98b009c3aca34d3be0d6781173301463735be8cbd4cae6af8a432be35ed59 +size 27837 diff --git a/completions/completions_00090.parquet b/completions/completions_00090.parquet new file mode 100644 index 0000000..49ef63c --- /dev/null +++ b/completions/completions_00090.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e21c4115af9243b8d8f0d61bfe6df669626efcbf58c42b35521a18191491ad0 +size 28696 diff --git a/completions/completions_00091.parquet b/completions/completions_00091.parquet new file mode 100644 index 0000000..212f998 --- /dev/null +++ b/completions/completions_00091.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e8a4b2fe8c071db2dfd37b5c03d74103bfbfe2cfa15e436b83edb242baa53e7c +size 29814 diff --git a/completions/completions_00092.parquet b/completions/completions_00092.parquet new file mode 100644 index 0000000..d74a691 --- /dev/null +++ b/completions/completions_00092.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e28409dcc77ab32ffae741ddaec48dae4ed2fbaf4e20afc027ad69091e24cb18 +size 27628 diff --git a/completions/completions_00093.parquet b/completions/completions_00093.parquet new file mode 100644 index 0000000..5ae2525 --- /dev/null +++ b/completions/completions_00093.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abe0546c6c2b6a7e5f2c623d5f44eb2ee9467e95c5e278380e39c4efa9b9deb5 +size 32621 diff --git a/completions/completions_00094.parquet b/completions/completions_00094.parquet new file mode 100644 index 0000000..1b3b7a1 --- /dev/null +++ b/completions/completions_00094.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:df987bf61614f196e8b238512046dfa615893bf1a698267a3385142d437d8f0b +size 22535 diff --git a/completions/completions_00095.parquet b/completions/completions_00095.parquet new file mode 100644 index 0000000..1301f18 --- /dev/null +++ b/completions/completions_00095.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4438f384973cf67dd16c0faa0edf5793e87b0ce420854953537d9d589c7e3a0d +size 30758 diff --git a/completions/completions_00096.parquet b/completions/completions_00096.parquet new file mode 100644 index 0000000..e5b64a3 --- /dev/null +++ b/completions/completions_00096.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f4e53012392b0ff07e1f17473a0fc36f54d7cbf70e55fded45629c8f2622d479 +size 31063 diff --git a/completions/completions_00097.parquet b/completions/completions_00097.parquet new file mode 100644 index 0000000..72955a7 --- /dev/null +++ b/completions/completions_00097.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b4ce78d03e5c9d59ac5a6ca8683993c894a9a31bbe01507331cb473d44cb783b +size 26911 diff --git a/completions/completions_00098.parquet b/completions/completions_00098.parquet new file mode 100644 index 0000000..8522cf0 --- /dev/null +++ b/completions/completions_00098.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:206a1d4c6f3d1a29f9b4e97d2c16c7e734589512fb214e31890883997ae4e151 +size 26935 diff --git a/completions/completions_00099.parquet b/completions/completions_00099.parquet new file mode 100644 index 0000000..b6f80ec --- /dev/null +++ b/completions/completions_00099.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:348ab8782ecd0060622a285b833c90aefb4c767e0658f43bc4349686d2e57e39 +size 26361 diff --git a/completions/completions_00100.parquet b/completions/completions_00100.parquet new file mode 100644 index 0000000..42048d8 --- /dev/null +++ b/completions/completions_00100.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b9e5c54b6bf63c592d2e6b0312eced32cbd50d50c8ed8ba2923e8ed74f65b59 +size 26273 diff --git a/completions/completions_00101.parquet b/completions/completions_00101.parquet new file mode 100644 index 0000000..be511b5 --- /dev/null +++ b/completions/completions_00101.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:395f4af23bbf613b3f17f2a1eb51b2e3083e11fe2945597c32c7fcf4df701868 +size 33384 diff --git a/completions/completions_00102.parquet b/completions/completions_00102.parquet new file mode 100644 index 0000000..649f837 --- /dev/null +++ b/completions/completions_00102.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d6dead04ed4652ccad7c18814e982fae5e8e8abf5da42b5e929184a4ab724262 +size 31834 diff --git a/completions/completions_00103.parquet b/completions/completions_00103.parquet new file mode 100644 index 0000000..8cff253 --- /dev/null +++ b/completions/completions_00103.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:342638f3bccc5304af3a002eacd2d6564b9084a72740b301de51ae2a53362dee +size 30557 diff --git a/completions/completions_00104.parquet b/completions/completions_00104.parquet new file mode 100644 index 0000000..2f2dd24 --- /dev/null +++ b/completions/completions_00104.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f2f66716604d215a6dfbd59035726cfd2bfa46ab3df4b05e7773c7fb86aa23b +size 30506 diff --git a/completions/completions_00105.parquet b/completions/completions_00105.parquet new file mode 100644 index 0000000..28408dc --- /dev/null +++ b/completions/completions_00105.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:79a3b562c36b8dc1ff7d7933ec83954c6aa2b2ab4ad56958e747173f05e69774 +size 31243 diff --git a/completions/completions_00106.parquet b/completions/completions_00106.parquet new file mode 100644 index 0000000..b5cc06e --- /dev/null +++ b/completions/completions_00106.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3d952cb9151662231d52616208761bd92a8e9d57850eaf19069fd844c18b5179 +size 27094 diff --git a/completions/completions_00107.parquet b/completions/completions_00107.parquet new file mode 100644 index 0000000..0375fff --- /dev/null +++ b/completions/completions_00107.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86eee6e1dd48ec4e6611aa914271b807bb1022497beb80a0becd4c925a0854ec +size 32037 diff --git a/completions/completions_00108.parquet b/completions/completions_00108.parquet new file mode 100644 index 0000000..e270afc --- /dev/null +++ b/completions/completions_00108.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10b455f2c44a252a3a9559aee4eed4414d4f34a5d31c5ce05079faf9f1686acc +size 26608 diff --git a/completions/completions_00109.parquet b/completions/completions_00109.parquet new file mode 100644 index 0000000..e112aba --- /dev/null +++ b/completions/completions_00109.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1c1d5f55bfec795154417485cf0c7e58c7883f098db69a3bf160a2e12da160ad +size 32684 diff --git a/completions/completions_00110.parquet b/completions/completions_00110.parquet new file mode 100644 index 0000000..40e16c7 --- /dev/null +++ b/completions/completions_00110.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9de50dcf8934017c474a04c48719bd7ada5ab2c99bfcf2dc744d198bef673cdb +size 31308 diff --git a/completions/completions_00111.parquet b/completions/completions_00111.parquet new file mode 100644 index 0000000..38b40be --- /dev/null +++ b/completions/completions_00111.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:732e3ae9863ddd910eb9d43dddcfdf290cc242f9c61001b02152761071a6951f +size 27648 diff --git a/completions/completions_00112.parquet b/completions/completions_00112.parquet new file mode 100644 index 0000000..f3b251f --- /dev/null +++ b/completions/completions_00112.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69adb415e0ae188b1606dd4af27fa16fa2d90823c612112d47a38d4875e1015d +size 31255 diff --git a/completions/completions_00113.parquet b/completions/completions_00113.parquet new file mode 100644 index 0000000..67199cd --- /dev/null +++ b/completions/completions_00113.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62e88dc7ce407a0bac296932df42ad7f284aaaf622191bf84178921cbb38143e +size 32187 diff --git a/completions/completions_00114.parquet b/completions/completions_00114.parquet new file mode 100644 index 0000000..177ab21 --- /dev/null +++ b/completions/completions_00114.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e796dfdc14bd2be31eff4a6d5069f918a7ee5b32e369abc99ff91e7391e8037 +size 34204 diff --git a/completions/completions_00115.parquet b/completions/completions_00115.parquet new file mode 100644 index 0000000..21e8944 --- /dev/null +++ b/completions/completions_00115.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f785c97a429f5c71f96315cc1d7aa589dc13ea08e706c1045e7df46d77fae5b8 +size 32344 diff --git a/completions/completions_00116.parquet b/completions/completions_00116.parquet new file mode 100644 index 0000000..06a1772 --- /dev/null +++ b/completions/completions_00116.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e05f869e273f8b97852b11b3e374976725c543d600aab9e06c59e02ea474b994 +size 33505 diff --git a/completions/completions_00117.parquet b/completions/completions_00117.parquet new file mode 100644 index 0000000..cea0a30 --- /dev/null +++ b/completions/completions_00117.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3d8e5587668d9911a6f475c4cb5f3f3041b70f66d03a0da7ce8c496b3e6bea43 +size 22914 diff --git a/completions/completions_00118.parquet b/completions/completions_00118.parquet new file mode 100644 index 0000000..f1b265c --- /dev/null +++ b/completions/completions_00118.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9d65e4b0dcdab772d36b8975b4988ca26ff25c90c5a7a3e176b27b0b48c4987e +size 32003 diff --git a/completions/completions_00119.parquet b/completions/completions_00119.parquet new file mode 100644 index 0000000..8d6a043 --- /dev/null +++ b/completions/completions_00119.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f431488fba7b5d2c5ab7eede58969de6b61e83ab373642b3230f3bc5c6757ee +size 28867 diff --git a/completions/completions_00120.parquet b/completions/completions_00120.parquet new file mode 100644 index 0000000..f33fb25 --- /dev/null +++ b/completions/completions_00120.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4d61d309248233d55eea2524d3aad7663b1b0c91f200c46a8fc258dd07d7610a +size 26970 diff --git a/completions/completions_00121.parquet b/completions/completions_00121.parquet new file mode 100644 index 0000000..bea190c --- /dev/null +++ b/completions/completions_00121.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9df30a34bab8bd7f79c73684d99b285c124518756f21d1ed3bd1ee55bbe45cf2 +size 28376 diff --git a/completions/completions_00122.parquet b/completions/completions_00122.parquet new file mode 100644 index 0000000..2f9d91d --- /dev/null +++ b/completions/completions_00122.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27383f2c29b11483e0bc2efaf60a9e15c4116873e2601b3b03a6a876d0902c52 +size 23368 diff --git a/completions/completions_00123.parquet b/completions/completions_00123.parquet new file mode 100644 index 0000000..883e38b --- /dev/null +++ b/completions/completions_00123.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dbb1156614a48b62c46f04348fef8e78f4d4dd5cc2d9d2228bfe4069da6f4c11 +size 27505 diff --git a/completions/completions_00124.parquet b/completions/completions_00124.parquet new file mode 100644 index 0000000..22ea1d3 --- /dev/null +++ b/completions/completions_00124.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e3d0e7410454163be916cbbcd95284250189fd73dd78611ef5e36e97d3b77cb7 +size 23223 diff --git a/completions/completions_00125.parquet b/completions/completions_00125.parquet new file mode 100644 index 0000000..4c722cf --- /dev/null +++ b/completions/completions_00125.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4daa53335ab5a96b61f96b697045621a9f3d96ff3ab3c4023560bfaa16bfc512 +size 23018 diff --git a/completions/completions_00126.parquet b/completions/completions_00126.parquet new file mode 100644 index 0000000..8dac454 --- /dev/null +++ b/completions/completions_00126.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bf4a6c9a729832b5292805b3ea9ce647c05f347b34e8a94f53c1673ac11e2fc7 +size 22731 diff --git a/completions/completions_00127.parquet b/completions/completions_00127.parquet new file mode 100644 index 0000000..66ca137 --- /dev/null +++ b/completions/completions_00127.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:920db6591f0ad5fcb728f5d3bd936e8d8e01e5136727e192848add14f8e25f18 +size 23382 diff --git a/completions/completions_00128.parquet b/completions/completions_00128.parquet new file mode 100644 index 0000000..914a02d --- /dev/null +++ b/completions/completions_00128.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:09242293912111186068399385dbc3067306b7b5474f73282d3a44c25c136696 +size 23371 diff --git a/completions/completions_00129.parquet b/completions/completions_00129.parquet new file mode 100644 index 0000000..82fac59 --- /dev/null +++ b/completions/completions_00129.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0ac4774230e9bcb4890896231013662296d03ae5eaf8c36dad2da4f64a4bc5b8 +size 27879 diff --git a/completions/completions_00130.parquet b/completions/completions_00130.parquet new file mode 100644 index 0000000..c6dfffa --- /dev/null +++ b/completions/completions_00130.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93181f532b21443eca3b863d007cb61914f5817e5f4344315788f0d9b84def69 +size 23119 diff --git a/completions/completions_00131.parquet b/completions/completions_00131.parquet new file mode 100644 index 0000000..45e5bf3 --- /dev/null +++ b/completions/completions_00131.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:97b904d2bdab62f42b4202969dca83661cf305687c21e821d53618132c4deb47 +size 23445 diff --git a/completions/completions_00132.parquet b/completions/completions_00132.parquet new file mode 100644 index 0000000..899c989 --- /dev/null +++ b/completions/completions_00132.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c9221a1eeb430490471c191d51a534f506480f8b9304ba98eb89d7b604556c5 +size 27704 diff --git a/completions/completions_00133.parquet b/completions/completions_00133.parquet new file mode 100644 index 0000000..11a156b --- /dev/null +++ b/completions/completions_00133.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7ac2e24c9878fe928d713932f0c76cd56b064832b1ab6f4ee2f0168564aee8ad +size 26954 diff --git a/completions/completions_00134.parquet b/completions/completions_00134.parquet new file mode 100644 index 0000000..c6eab07 --- /dev/null +++ b/completions/completions_00134.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:af5213e448593b599904acbce6140547bb943a51e827f7221f0fd82551f711c1 +size 26561 diff --git a/completions/completions_00135.parquet b/completions/completions_00135.parquet new file mode 100644 index 0000000..6dcc669 --- /dev/null +++ b/completions/completions_00135.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:da82b146504dace30e5ccc4e147d53b709a85538b8aedfa6b4849a61a8ee32a6 +size 27626 diff --git a/completions/completions_00136.parquet b/completions/completions_00136.parquet new file mode 100644 index 0000000..17a9e6c --- /dev/null +++ b/completions/completions_00136.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3f1c94d3caaa8e6ff8f938d19e6378c0f6bc975c517d5233966c344f8a1f7bf4 +size 32497 diff --git a/completions/completions_00137.parquet b/completions/completions_00137.parquet new file mode 100644 index 0000000..75ccc4e --- /dev/null +++ b/completions/completions_00137.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:887387af68fea0ca6dad5c503bb5420d44f7b8a18a37a8278eea3df491082339 +size 22802 diff --git a/completions/completions_00138.parquet b/completions/completions_00138.parquet new file mode 100644 index 0000000..702ac0e --- /dev/null +++ b/completions/completions_00138.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:81d6200d3d161dd923eee4b5b2d5f599a2db3a5f281be4ea0d9dda44600376cc +size 27292 diff --git a/completions/completions_00139.parquet b/completions/completions_00139.parquet new file mode 100644 index 0000000..31383b3 --- /dev/null +++ b/completions/completions_00139.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6416980d7722337eccd7cd5ce84bc6f8ac2278c9d44bbfa282c187ed772cc385 +size 27426 diff --git a/completions/completions_00140.parquet b/completions/completions_00140.parquet new file mode 100644 index 0000000..be0a427 --- /dev/null +++ b/completions/completions_00140.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53e593c2a49500f0e6394d44b0a4bddcbc4b865a802cfb33f1b16a77ed820cfe +size 27080 diff --git a/completions/completions_00141.parquet b/completions/completions_00141.parquet new file mode 100644 index 0000000..210654b --- /dev/null +++ b/completions/completions_00141.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:969d6b1586b82669a0e5963a3aa531bfec6620e988858e999eb2a963eeba0763 +size 26695 diff --git a/completions/completions_00142.parquet b/completions/completions_00142.parquet new file mode 100644 index 0000000..ae6caaf --- /dev/null +++ b/completions/completions_00142.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f983ca3798e66c75d399bfdf3c27daac552c80e1d7ee098b3bd2014278530718 +size 26596 diff --git a/completions/completions_00143.parquet b/completions/completions_00143.parquet new file mode 100644 index 0000000..8b52404 --- /dev/null +++ b/completions/completions_00143.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04a94cb44d6d5e76af25aa47c027d08f3b0e1e023befea55e82bd1c194613f2e +size 22687 diff --git a/completions/completions_00144.parquet b/completions/completions_00144.parquet new file mode 100644 index 0000000..d877981 --- /dev/null +++ b/completions/completions_00144.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0dade6f94d1551969d60a8f8811924ee4af8aea8ee2156dc6174874b98252377 +size 26791 diff --git a/completions/completions_00145.parquet b/completions/completions_00145.parquet new file mode 100644 index 0000000..6e440de --- /dev/null +++ b/completions/completions_00145.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:32ab8ed79123e22f16884fa7648fad9aba40cc3bfa0010089cdfb315dcfb1801 +size 32495 diff --git a/completions/completions_00146.parquet b/completions/completions_00146.parquet new file mode 100644 index 0000000..cbc49b9 --- /dev/null +++ b/completions/completions_00146.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5b7e2312816e6badb16d806250d979bd24b973f6a5356ef01aa5d5a2e146d37d +size 32251 diff --git a/completions/completions_00147.parquet b/completions/completions_00147.parquet new file mode 100644 index 0000000..c9363ba --- /dev/null +++ b/completions/completions_00147.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77f87ef922370e2d54d08872bd97dd251087dd498d63840538a0131ecaa3776d +size 28876 diff --git a/completions/completions_00148.parquet b/completions/completions_00148.parquet new file mode 100644 index 0000000..32b4657 --- /dev/null +++ b/completions/completions_00148.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:08c46752507932442bcc61d14e77afecd303222edbe789701ba0b11c219838ce +size 31215 diff --git a/completions/completions_00149.parquet b/completions/completions_00149.parquet new file mode 100644 index 0000000..78a40f3 --- /dev/null +++ b/completions/completions_00149.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47d2de0538425c8f166ee87ea24026c9bd00e3572be89030412faa10d0b23696 +size 27971 diff --git a/completions/completions_00150.parquet b/completions/completions_00150.parquet new file mode 100644 index 0000000..9923c90 --- /dev/null +++ b/completions/completions_00150.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:44d336da79f4797c5be9ebe05ab4e6f3866aca2ffbc716404b33b2266f12d98a +size 34117 diff --git a/completions/completions_00151.parquet b/completions/completions_00151.parquet new file mode 100644 index 0000000..27e153f --- /dev/null +++ b/completions/completions_00151.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4a3b982c5c2485d1f752c3a4a560e5963af77fbc32dc2cec3ea2ccd55c5cea32 +size 27225 diff --git a/completions/completions_00152.parquet b/completions/completions_00152.parquet new file mode 100644 index 0000000..cd4c792 --- /dev/null +++ b/completions/completions_00152.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:897cfa45d7528de7e4badeb40043bf2e3b83d3ae11ef3deb9fe8240282692830 +size 26260 diff --git a/completions/completions_00153.parquet b/completions/completions_00153.parquet new file mode 100644 index 0000000..23ea5db --- /dev/null +++ b/completions/completions_00153.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:499cebe1aee748f2f448c7cd406c7f19b4121032a9a0748e22cd928317b410fc +size 28149 diff --git a/completions/completions_00154.parquet b/completions/completions_00154.parquet new file mode 100644 index 0000000..7c1b443 --- /dev/null +++ b/completions/completions_00154.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d6805599b832d38e8754349a720b62f4b9dacd089fb290fa88792afe246c369b +size 28459 diff --git a/completions/completions_00155.parquet b/completions/completions_00155.parquet new file mode 100644 index 0000000..b6d0546 --- /dev/null +++ b/completions/completions_00155.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04421c35f234ca2e504a0317a187ffb9fd1c8f962d4f2e5edc4bad8862f8868f +size 27904 diff --git a/completions/completions_00156.parquet b/completions/completions_00156.parquet new file mode 100644 index 0000000..3bed26e --- /dev/null +++ b/completions/completions_00156.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5da608d3dccc5cbf8307420bae4b7078d4098e5bb976912483c607e5af5507a0 +size 27792 diff --git a/completions/completions_00157.parquet b/completions/completions_00157.parquet new file mode 100644 index 0000000..0caeffd --- /dev/null +++ b/completions/completions_00157.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d860e1305d202d85d4eb6839b51d5220ba54166a7e98c09c7c37ad8455f1b643 +size 29608 diff --git a/completions/completions_00158.parquet b/completions/completions_00158.parquet new file mode 100644 index 0000000..1d4da87 --- /dev/null +++ b/completions/completions_00158.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e3d168b97fc1d3d295fc316823e14ece6fa1516604acf51a29c5435b3661f260 +size 27391 diff --git a/completions/completions_00159.parquet b/completions/completions_00159.parquet new file mode 100644 index 0000000..203d33c --- /dev/null +++ b/completions/completions_00159.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f089b6e4432ccffbf6a96220e6479a61724c17efb4d823a6775268e21d99e1a9 +size 26321 diff --git a/completions/completions_00160.parquet b/completions/completions_00160.parquet new file mode 100644 index 0000000..e925b54 --- /dev/null +++ b/completions/completions_00160.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5039c4024a4e12dfd9926421d5371888037a8e52e8a939f620a12ec212e12111 +size 27198 diff --git a/completions/completions_00161.parquet b/completions/completions_00161.parquet new file mode 100644 index 0000000..5fc088f --- /dev/null +++ b/completions/completions_00161.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2a44015993132128354a5a5c7af6d7a27807c5453e7631f5d9425c4942a0fd4b +size 30947 diff --git a/completions/completions_00162.parquet b/completions/completions_00162.parquet new file mode 100644 index 0000000..427ee33 --- /dev/null +++ b/completions/completions_00162.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4c4ccd0a651d284eb967089cae49753ce20ee221fb6ae8ebb565b5db3f8f23b +size 33058 diff --git a/completions/completions_00163.parquet b/completions/completions_00163.parquet new file mode 100644 index 0000000..c531ff5 --- /dev/null +++ b/completions/completions_00163.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f3aeca89cfb1744fc37fb603844c1246b1cb1155d167a59a4d7d25c0d5cea8aa +size 28124 diff --git a/completions/completions_00164.parquet b/completions/completions_00164.parquet new file mode 100644 index 0000000..48543bb --- /dev/null +++ b/completions/completions_00164.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:46207e758642df8246e3913807ca29f2d095e68ce504fde75155b65d980ab61a +size 28182 diff --git a/completions/completions_00165.parquet b/completions/completions_00165.parquet new file mode 100644 index 0000000..c5055a7 --- /dev/null +++ b/completions/completions_00165.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b60f25ff3d75e666c31a7c9823e2a68163ea32103f25ba4e700f87fb8cb8736f +size 28372 diff --git a/completions/completions_00166.parquet b/completions/completions_00166.parquet new file mode 100644 index 0000000..8ffa04c --- /dev/null +++ b/completions/completions_00166.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1866ff9c04c65fef3bfb94b87636ebfad6ac5ef07c56600b401e264eaf9c4162 +size 26655 diff --git a/completions/completions_00167.parquet b/completions/completions_00167.parquet new file mode 100644 index 0000000..b321d6e --- /dev/null +++ b/completions/completions_00167.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8176cbab90aaf7625a3f5b75b93739045553c164a34c66138f54aee6595d987d +size 29067 diff --git a/completions/completions_00168.parquet b/completions/completions_00168.parquet new file mode 100644 index 0000000..4e6cb69 --- /dev/null +++ b/completions/completions_00168.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7fbe6f52941ceb5987155dac9a7c4526a1e4dad00529ee152e0e34ffff22c11 +size 28096 diff --git a/completions/completions_00169.parquet b/completions/completions_00169.parquet new file mode 100644 index 0000000..c2c0736 --- /dev/null +++ b/completions/completions_00169.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f44e73f8b082ad8458e47e3b10cfb9716570ed0ec8982c6af70ad9fcec61d8d4 +size 27474 diff --git a/completions/completions_00170.parquet b/completions/completions_00170.parquet new file mode 100644 index 0000000..9f53198 --- /dev/null +++ b/completions/completions_00170.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:930db466780c76505b3fac7b87dd0311649837f677e10e77d01b114021f5e5fc +size 23231 diff --git a/completions/completions_00171.parquet b/completions/completions_00171.parquet new file mode 100644 index 0000000..ff24d4f --- /dev/null +++ b/completions/completions_00171.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:912752524330bd4e4935a389976c7f923c39cb26416d763b44f69cb9a5f216d9 +size 22937 diff --git a/completions/completions_00172.parquet b/completions/completions_00172.parquet new file mode 100644 index 0000000..4b93e01 --- /dev/null +++ b/completions/completions_00172.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12c1108b23f25ffb823976420997360bd850b26cb672a36bcc67b9acc358616d +size 23602 diff --git a/completions/completions_00173.parquet b/completions/completions_00173.parquet new file mode 100644 index 0000000..60bad1a --- /dev/null +++ b/completions/completions_00173.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad37687e1c8133112ec9eb96f1562f8966cbdb2708982e924110525b6590cb4f +size 23093 diff --git a/completions/completions_00174.parquet b/completions/completions_00174.parquet new file mode 100644 index 0000000..7befc88 --- /dev/null +++ b/completions/completions_00174.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:58b011562f924e7de002c48f4041338502ee2f9b40f5efe2cab97210c026b240 +size 29016 diff --git a/completions/completions_00175.parquet b/completions/completions_00175.parquet new file mode 100644 index 0000000..ab2c185 --- /dev/null +++ b/completions/completions_00175.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b69d856bf4caef0fe843c1e645de46949f2d8554c1a31d9e9cfe9e8de48153e5 +size 22681 diff --git a/completions/completions_00176.parquet b/completions/completions_00176.parquet new file mode 100644 index 0000000..edc2790 --- /dev/null +++ b/completions/completions_00176.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b189c1dd864319bc9130deb36ea474d45fb8c5b45781ce294e06aef79969dcc8 +size 22895 diff --git a/completions/completions_00177.parquet b/completions/completions_00177.parquet new file mode 100644 index 0000000..b5e875d --- /dev/null +++ b/completions/completions_00177.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c13e7fa345854a1a7dc19537549c426918d9a18258c001df55ba583a23076015 +size 27652 diff --git a/completions/completions_00178.parquet b/completions/completions_00178.parquet new file mode 100644 index 0000000..138cc3f --- /dev/null +++ b/completions/completions_00178.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:195f14ef19bb611886bb3952248ad5f14a5afafcb8f5acd6f9ffee25744247e5 +size 23091 diff --git a/completions/completions_00179.parquet b/completions/completions_00179.parquet new file mode 100644 index 0000000..8f15270 --- /dev/null +++ b/completions/completions_00179.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7d1065bbc630bb339f11bf71e1790a6420570a778c38e5bc9dca230389e36f3 +size 29676 diff --git a/completions/completions_00180.parquet b/completions/completions_00180.parquet new file mode 100644 index 0000000..7f8e780 --- /dev/null +++ b/completions/completions_00180.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:99795fc9118fd48ffd0deb2df270a8734a715b4339c3fd632cc39df7c2c4ffaf +size 28392 diff --git a/completions/completions_00181.parquet b/completions/completions_00181.parquet new file mode 100644 index 0000000..c5c3a68 --- /dev/null +++ b/completions/completions_00181.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5616f4a10707022fde23687d04872e124acb58150692db4e6edc2406fdfb53e1 +size 23215 diff --git a/completions/completions_00182.parquet b/completions/completions_00182.parquet new file mode 100644 index 0000000..87631d8 --- /dev/null +++ b/completions/completions_00182.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2276a26978f54aeeb597df6626e7bb819e6cd47fdbe0e01d3f6ce3c45dafc486 +size 23039 diff --git a/completions/completions_00183.parquet b/completions/completions_00183.parquet new file mode 100644 index 0000000..2eadd93 --- /dev/null +++ b/completions/completions_00183.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a89ea852355decc7f7e9bf39390cd36b886d693f9ae4d5bf37a0062d6b2a1024 +size 23529 diff --git a/completions/completions_00184.parquet b/completions/completions_00184.parquet new file mode 100644 index 0000000..74c10bd --- /dev/null +++ b/completions/completions_00184.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4eee4ae009f899c849b98403764cdf4b615d4e7f8f6b7a1c594ea0587ca1cae +size 31371 diff --git a/completions/completions_00185.parquet b/completions/completions_00185.parquet new file mode 100644 index 0000000..18f352b --- /dev/null +++ b/completions/completions_00185.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3d3c82b753b3f15fd3d99cd242bd09fea9ad43828fa0878a398d70bb8828675f +size 35975 diff --git a/completions/completions_00186.parquet b/completions/completions_00186.parquet new file mode 100644 index 0000000..fbbcf82 --- /dev/null +++ b/completions/completions_00186.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:852b5b433d1b0661143b78cf6c2e9042ddccc6f3f9498704995d0112f7b10c54 +size 28821 diff --git a/completions/completions_00187.parquet b/completions/completions_00187.parquet new file mode 100644 index 0000000..52a7264 --- /dev/null +++ b/completions/completions_00187.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7368377739c22f217607cb4db77c8b1af1ecd603a0c78136a54827c59e6c3ae9 +size 28182 diff --git a/completions/completions_00188.parquet b/completions/completions_00188.parquet new file mode 100644 index 0000000..5645619 --- /dev/null +++ b/completions/completions_00188.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8929665db7675bfc8e8e3dd9064a9fe047a6ac7f0351f39a2b3b595256fcd273 +size 23520 diff --git a/completions/completions_00189.parquet b/completions/completions_00189.parquet new file mode 100644 index 0000000..f74cb26 --- /dev/null +++ b/completions/completions_00189.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad321e0c43248e4dc645f72f9941609026c3e8d8a4e3f07a6e794ad5db928a10 +size 26920 diff --git a/completions/completions_00190.parquet b/completions/completions_00190.parquet new file mode 100644 index 0000000..5355f15 --- /dev/null +++ b/completions/completions_00190.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:714de2f212a99b6f909cce991b54c35d3ef67116995ba41bc326333cd6c8b5e5 +size 33011 diff --git a/completions/completions_00191.parquet b/completions/completions_00191.parquet new file mode 100644 index 0000000..1516ba0 --- /dev/null +++ b/completions/completions_00191.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:87ba9a4035870b62fac58c1f393feb480bf763f0e77ca577b46d309054d447df +size 23128 diff --git a/completions/completions_00192.parquet b/completions/completions_00192.parquet new file mode 100644 index 0000000..da4c939 --- /dev/null +++ b/completions/completions_00192.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:af974830920fc76ccecdc16f0982691d1899b15a3fd79227d3f039ee6b154726 +size 32322 diff --git a/completions/completions_00193.parquet b/completions/completions_00193.parquet new file mode 100644 index 0000000..585e8a4 --- /dev/null +++ b/completions/completions_00193.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fb8e16cd4f7a7ac04eb5d44bc7006dbee1fb5a6007285a0b7d0713d2d53ca362 +size 30962 diff --git a/completions/completions_00194.parquet b/completions/completions_00194.parquet new file mode 100644 index 0000000..93edbb4 --- /dev/null +++ b/completions/completions_00194.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3afea69ab2bd707cd189db3035c27155e09f6fa9d67e6f1d2669fe180f207767 +size 28253 diff --git a/completions/completions_00195.parquet b/completions/completions_00195.parquet new file mode 100644 index 0000000..500074d --- /dev/null +++ b/completions/completions_00195.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0fa3464ecfd352ece423dbaadf2901d28901ffcc4e0ca8b74ed128aa5f36c557 +size 33086 diff --git a/completions/completions_00196.parquet b/completions/completions_00196.parquet new file mode 100644 index 0000000..5c9553e --- /dev/null +++ b/completions/completions_00196.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5889bbba1c976521afb38addde852a16017f9a040ddccd9643310dcaef27bde9 +size 26672 diff --git a/completions/completions_00197.parquet b/completions/completions_00197.parquet new file mode 100644 index 0000000..8b66d77 --- /dev/null +++ b/completions/completions_00197.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b30ceee6db964ce26a8a993354cf2f79143d3b84a354327ffe22eeb9022e8fcf +size 29903 diff --git a/completions/completions_00198.parquet b/completions/completions_00198.parquet new file mode 100644 index 0000000..4e61eaa --- /dev/null +++ b/completions/completions_00198.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:877dee937057d056a01b395b144643d5db9529ef2e718c89abe8a570ad9cf76b +size 32639 diff --git a/completions/completions_00199.parquet b/completions/completions_00199.parquet new file mode 100644 index 0000000..245e66e --- /dev/null +++ b/completions/completions_00199.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1167ec085f313b2b469a379435e3fc950db1702299f43082e64780e5fe9443f1 +size 32669 diff --git a/completions/completions_00200.parquet b/completions/completions_00200.parquet new file mode 100644 index 0000000..706f27f --- /dev/null +++ b/completions/completions_00200.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fb6e748a34e05e856e031aaee4398c9bebfa536f6b9f637443cb3606b31a1b2c +size 27189 diff --git a/completions/completions_00201.parquet b/completions/completions_00201.parquet new file mode 100644 index 0000000..7167c73 --- /dev/null +++ b/completions/completions_00201.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f462215f123278ea83ea534ad0911a216f8a0fdcd2f486a15941ac1d8462f0f3 +size 28481 diff --git a/completions/completions_00202.parquet b/completions/completions_00202.parquet new file mode 100644 index 0000000..6cc4a83 --- /dev/null +++ b/completions/completions_00202.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02b787d48008e31b42bdfe69bb4d914cc9d6a813b1e22cbbab36928b6f02d4ff +size 27380 diff --git a/completions/completions_00203.parquet b/completions/completions_00203.parquet new file mode 100644 index 0000000..76875ab --- /dev/null +++ b/completions/completions_00203.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6e46f9f1458af1e486281dd403e182ed087439bef7088c2a206fdb7468c6f02c +size 28325 diff --git a/completions/completions_00204.parquet b/completions/completions_00204.parquet new file mode 100644 index 0000000..fe1cdd7 --- /dev/null +++ b/completions/completions_00204.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eb51312f46ba43a7d95545b490dcea2cd7c85fedfeb4cbc83099b4080a037d5b +size 30645 diff --git a/completions/completions_00205.parquet b/completions/completions_00205.parquet new file mode 100644 index 0000000..64be447 --- /dev/null +++ b/completions/completions_00205.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21579037e69569895de35fe83ce39c042097264ae4a863e1efc3af976030289d +size 31823 diff --git a/completions/completions_00206.parquet b/completions/completions_00206.parquet new file mode 100644 index 0000000..f475a77 --- /dev/null +++ b/completions/completions_00206.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2f45dd108341c42111eb5248a36c0d389e4f7574ee1fa031b5942158d8cd181e +size 30185 diff --git a/completions/completions_00207.parquet b/completions/completions_00207.parquet new file mode 100644 index 0000000..4b2080b --- /dev/null +++ b/completions/completions_00207.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ece22bfe9c1b46ae3c5e8b103d24c9099cb9f97a42750f142913c1c2c0a6e427 +size 34234 diff --git a/completions/completions_00208.parquet b/completions/completions_00208.parquet new file mode 100644 index 0000000..51f7af8 --- /dev/null +++ b/completions/completions_00208.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:becca6c0c2441800a45f23aef7126e83e184452e219f4cc701074951bd443239 +size 27788 diff --git a/completions/completions_00209.parquet b/completions/completions_00209.parquet new file mode 100644 index 0000000..c0e2e19 --- /dev/null +++ b/completions/completions_00209.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3656ae002146c81ff4558daad1650e7d43f5c3c66e26cacd1cd0fc5df0dd5f13 +size 32707 diff --git a/completions/completions_00210.parquet b/completions/completions_00210.parquet new file mode 100644 index 0000000..2c17b60 --- /dev/null +++ b/completions/completions_00210.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27216f7297f1fe528a93409d8a772a0b39b807c8f3441fed9a1e4388e1b21307 +size 33145 diff --git a/completions/completions_00211.parquet b/completions/completions_00211.parquet new file mode 100644 index 0000000..16ab80b --- /dev/null +++ b/completions/completions_00211.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0be9beb6874a90db20a0cbbadf9650bc00ba36ebd1d3caac8fa8a7e8091b3b8c +size 31438 diff --git a/completions/completions_00212.parquet b/completions/completions_00212.parquet new file mode 100644 index 0000000..d1cc353 --- /dev/null +++ b/completions/completions_00212.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b14a43e382a392630e06c74aa0b01aeaad386a00e96c9bf5dd9b8c8e4d30aa5 +size 32830 diff --git a/completions/completions_00213.parquet b/completions/completions_00213.parquet new file mode 100644 index 0000000..c6a57eb --- /dev/null +++ b/completions/completions_00213.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9b89204541f8f2982f5a724f83852425ab20778fee8fecb4b8dc681ab9ed8ba5 +size 27570 diff --git a/completions/completions_00214.parquet b/completions/completions_00214.parquet new file mode 100644 index 0000000..4865514 --- /dev/null +++ b/completions/completions_00214.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:84b5047f961f0b6bc8dc9c787139e15609d44b29b5e846acb6c3572f4087f011 +size 26644 diff --git a/completions/completions_00215.parquet b/completions/completions_00215.parquet new file mode 100644 index 0000000..9bfcea1 --- /dev/null +++ b/completions/completions_00215.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:41950a0c02138750a7218eec1d73fb9656b5f4a9c2dc58e94c7371480b5251d9 +size 27929 diff --git a/completions/completions_00216.parquet b/completions/completions_00216.parquet new file mode 100644 index 0000000..42f214b --- /dev/null +++ b/completions/completions_00216.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:601e352f4c76b936787711c69decda7fc24185fd4b3b24c945204b8336736179 +size 27796 diff --git a/completions/completions_00217.parquet b/completions/completions_00217.parquet new file mode 100644 index 0000000..456cc33 --- /dev/null +++ b/completions/completions_00217.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6e1b9d48aee23773957c39f45c4a4444bd10b0aff957158954c121795357112a +size 28279 diff --git a/completions/completions_00218.parquet b/completions/completions_00218.parquet new file mode 100644 index 0000000..e548e36 --- /dev/null +++ b/completions/completions_00218.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a12e3349028820a1958d2cea30e0fa8979dd1ea9c23c62904fb892feb78c8ce3 +size 31509 diff --git a/completions/completions_00219.parquet b/completions/completions_00219.parquet new file mode 100644 index 0000000..8b6dc7d --- /dev/null +++ b/completions/completions_00219.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bc95c21f2884d5f5604d970f26d0032491bd3c8c35200495ddf7ca40b72cd3e4 +size 26884 diff --git a/completions/completions_00220.parquet b/completions/completions_00220.parquet new file mode 100644 index 0000000..7a7f7c9 --- /dev/null +++ b/completions/completions_00220.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:64b196c216ba954a95ff24d9e0c16081ca31f1c83b5e6b7770e8005bdd419cd3 +size 31067 diff --git a/completions/completions_00221.parquet b/completions/completions_00221.parquet new file mode 100644 index 0000000..8d83648 --- /dev/null +++ b/completions/completions_00221.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9d70f4047cc66cd63fe893a15160ad17c435bfcb7d91b32f75a04f757dd8f203 +size 28195 diff --git a/completions/completions_00222.parquet b/completions/completions_00222.parquet new file mode 100644 index 0000000..36d1a8f --- /dev/null +++ b/completions/completions_00222.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:28bd3d92fbbc2dd6630839f7c9fb043580324c1d87b95fd45cb4c1f42268788d +size 27879 diff --git a/completions/completions_00223.parquet b/completions/completions_00223.parquet new file mode 100644 index 0000000..e02e6bb --- /dev/null +++ b/completions/completions_00223.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76866d832876779e795c4c4b77dc0e7d682d8c6d4b319d623dd31eb802fbd406 +size 28329 diff --git a/completions/completions_00224.parquet b/completions/completions_00224.parquet new file mode 100644 index 0000000..a7a9e2b --- /dev/null +++ b/completions/completions_00224.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9d0b183dc7879b11e5d7a70a8cb490844994598405cfa1a5c4e58d338bb7137 +size 27226 diff --git a/completions/completions_00225.parquet b/completions/completions_00225.parquet new file mode 100644 index 0000000..a300d42 --- /dev/null +++ b/completions/completions_00225.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62ddb73462cbc435c354586a82e589f1b3b8fb9db1ca40fc36b6113c3caa3a7c +size 27698 diff --git a/completions/completions_00226.parquet b/completions/completions_00226.parquet new file mode 100644 index 0000000..02a5627 --- /dev/null +++ b/completions/completions_00226.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:108fba56adc724e97a3c142cf8ece321e74ff8bc976a057573e10b0c335aa6a7 +size 23217 diff --git a/completions/completions_00227.parquet b/completions/completions_00227.parquet new file mode 100644 index 0000000..6f4f100 --- /dev/null +++ b/completions/completions_00227.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:05ce1cafff836a30d62c0ac9d0ffb079baff412cf89075bd0ebb406cb4ab745d +size 28439 diff --git a/completions/completions_00228.parquet b/completions/completions_00228.parquet new file mode 100644 index 0000000..3859271 --- /dev/null +++ b/completions/completions_00228.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d9548f74fa7d14190c04f7c6e5137f5e3c2bed1a7f6cd4e3f9de740da55861d +size 31676 diff --git a/completions/completions_00229.parquet b/completions/completions_00229.parquet new file mode 100644 index 0000000..82fa4c9 --- /dev/null +++ b/completions/completions_00229.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c909b2a457daade3b16549674af61b3d943feccbd2dae928405e0aab2123115 +size 22992 diff --git a/completions/completions_00230.parquet b/completions/completions_00230.parquet new file mode 100644 index 0000000..5c51550 --- /dev/null +++ b/completions/completions_00230.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:32038ccb2a58f66b13447e5dad90fda3ecda48363276ac33af3ad58bfd0237ca +size 34090 diff --git a/completions/completions_00231.parquet b/completions/completions_00231.parquet new file mode 100644 index 0000000..2e1a2e0 --- /dev/null +++ b/completions/completions_00231.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:59284cc8032afacde01f0bf21f6e03710653ab6059d5efdba7d851a74af1d5dc +size 31847 diff --git a/completions/completions_00232.parquet b/completions/completions_00232.parquet new file mode 100644 index 0000000..55ffc63 --- /dev/null +++ b/completions/completions_00232.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:883a0d37d4ca3a3f3724926ab6b0a0c37992491e38a35a8cae3001d773d8d684 +size 22875 diff --git a/completions/completions_00233.parquet b/completions/completions_00233.parquet new file mode 100644 index 0000000..b8fca52 --- /dev/null +++ b/completions/completions_00233.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:813de7efec8e8658aad13d29ef4cc121d404e836965470bf85586498e16b9c7f +size 33807 diff --git a/completions/completions_00234.parquet b/completions/completions_00234.parquet new file mode 100644 index 0000000..aa0961e --- /dev/null +++ b/completions/completions_00234.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77460fbe50aa5b1b4bd2da4ddd436f32a7f6357f6fe3b3e5ef09ce7bc95a0d41 +size 32846 diff --git a/completions/completions_00235.parquet b/completions/completions_00235.parquet new file mode 100644 index 0000000..73c5b93 --- /dev/null +++ b/completions/completions_00235.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:296a5f381ab95a536215332b13a0da2dae48511772a226866f7bc1caf4c3c0f2 +size 28501 diff --git a/completions/completions_00236.parquet b/completions/completions_00236.parquet new file mode 100644 index 0000000..d1fed74 --- /dev/null +++ b/completions/completions_00236.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b1f35290f052a020d94bf825a3b541b195920c4001c9168e7a3e44d69afed9b +size 32196 diff --git a/completions/completions_00237.parquet b/completions/completions_00237.parquet new file mode 100644 index 0000000..a7c84c8 --- /dev/null +++ b/completions/completions_00237.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:81da42b8297b9c30dab31b7faa5164dcf3927cf0bb15729bff1af0600c51fcb7 +size 27319 diff --git a/completions/completions_00238.parquet b/completions/completions_00238.parquet new file mode 100644 index 0000000..4ef1ff1 --- /dev/null +++ b/completions/completions_00238.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e32958948b49f4bd2ad226cbbe81f7d781b3497fec503bf4e52984dad4ee3689 +size 27393 diff --git a/completions/completions_00239.parquet b/completions/completions_00239.parquet new file mode 100644 index 0000000..6a6be32 --- /dev/null +++ b/completions/completions_00239.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d153282fda137c94863407575472755c7e30525603ea9ac52e3512e021dade7 +size 23741 diff --git a/completions/completions_00240.parquet b/completions/completions_00240.parquet new file mode 100644 index 0000000..fc9e61f --- /dev/null +++ b/completions/completions_00240.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3a8d92e7fbee95663db8bd7122abac3edfd5c37bed7ae437daccdc4e29066fe6 +size 32720 diff --git a/completions/completions_00241.parquet b/completions/completions_00241.parquet new file mode 100644 index 0000000..af41102 --- /dev/null +++ b/completions/completions_00241.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be1b978ccb3812fc9ad9b78c8eefc28678b103a496ad65013a2831dc1dfadf8d +size 31904 diff --git a/completions/completions_00242.parquet b/completions/completions_00242.parquet new file mode 100644 index 0000000..9846ffd --- /dev/null +++ b/completions/completions_00242.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6bdc0cd9e338a967e82dea6eeff04f3dc53b32be8c7a7ef7b3fc7aff1062015 +size 29456 diff --git a/completions/completions_00243.parquet b/completions/completions_00243.parquet new file mode 100644 index 0000000..24f5acc --- /dev/null +++ b/completions/completions_00243.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:539e415613e4bc9168fbf83f8aa01e5de94346481f660e2674aa7b27b1dcfd03 +size 35175 diff --git a/completions/completions_00244.parquet b/completions/completions_00244.parquet new file mode 100644 index 0000000..c4f3271 --- /dev/null +++ b/completions/completions_00244.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc0e67867fe4179c95915be715a2c114d021f670beb8a9c483163e6263ffe6a6 +size 28412 diff --git a/completions/completions_00245.parquet b/completions/completions_00245.parquet new file mode 100644 index 0000000..16b7abe --- /dev/null +++ b/completions/completions_00245.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7dd8f007fce078f99611491b5aea76af89b9afaabd371444917df93419555965 +size 31326 diff --git a/completions/completions_00246.parquet b/completions/completions_00246.parquet new file mode 100644 index 0000000..17cd485 --- /dev/null +++ b/completions/completions_00246.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3aaa48e966784583f6894cb6fda820b76397c750145535d5189248e0369494e8 +size 32584 diff --git a/completions/completions_00247.parquet b/completions/completions_00247.parquet new file mode 100644 index 0000000..7605691 --- /dev/null +++ b/completions/completions_00247.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:441e47abb05a075af4a04927a42c4afbd3ef7c5bef125a4e8073b571a5da6c6e +size 32518 diff --git a/completions/completions_00248.parquet b/completions/completions_00248.parquet new file mode 100644 index 0000000..54d4c23 --- /dev/null +++ b/completions/completions_00248.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:106bb0f10401cae0ee2d00295868e934d90624437835af5e08ba804b0af20141 +size 28524 diff --git a/completions/completions_00249.parquet b/completions/completions_00249.parquet new file mode 100644 index 0000000..8ab4ecf --- /dev/null +++ b/completions/completions_00249.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f26fd3c15e3aaa1e57a027b8d3773a76375ce203ee9c4e718b06828dc738d7eb +size 23031 diff --git a/completions/completions_00250.parquet b/completions/completions_00250.parquet new file mode 100644 index 0000000..fe9b06c --- /dev/null +++ b/completions/completions_00250.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:16b12496784e18b7d69e916161ef83bd17df5e1e107def6d048ed8936a5e7f1b +size 31711 diff --git a/completions/completions_00251.parquet b/completions/completions_00251.parquet new file mode 100644 index 0000000..98ed306 --- /dev/null +++ b/completions/completions_00251.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cbf9d9658b1dd6decb13622c443d0889534a2f223c75181edeacfb8b03523a6d +size 27100 diff --git a/completions/completions_00252.parquet b/completions/completions_00252.parquet new file mode 100644 index 0000000..e5b1226 --- /dev/null +++ b/completions/completions_00252.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a62b9836879b6c668ed431c643b2dbb08be1be090cad7b5d52b61f2938f4e865 +size 28519 diff --git a/completions/completions_00253.parquet b/completions/completions_00253.parquet new file mode 100644 index 0000000..0c35091 --- /dev/null +++ b/completions/completions_00253.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1973cf1ebec384e1fa135f8b79269bd4d2dffd1429fe3775e68598961e8dcb29 +size 27129 diff --git a/completions/completions_00254.parquet b/completions/completions_00254.parquet new file mode 100644 index 0000000..efd8ce1 --- /dev/null +++ b/completions/completions_00254.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc7a90180cea27fe2a2ff0aa9fa2e3cf60676e9f377fd4db24c1fa3c5bc0b475 +size 31969 diff --git a/completions/completions_00255.parquet b/completions/completions_00255.parquet new file mode 100644 index 0000000..185fc94 --- /dev/null +++ b/completions/completions_00255.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a141c3bed9bcbb70e2a814ed48c2f4acbe3853b2a54e53b77566f62a38aec6a +size 28108 diff --git a/completions/completions_00256.parquet b/completions/completions_00256.parquet new file mode 100644 index 0000000..52aff43 --- /dev/null +++ b/completions/completions_00256.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3be38230e16444404560500f12dd0030ae790e0438c0ea09441e918db9399a93 +size 22364 diff --git a/completions/completions_00257.parquet b/completions/completions_00257.parquet new file mode 100644 index 0000000..6f116fd --- /dev/null +++ b/completions/completions_00257.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cf9729455a075259359403678575a2ed10a74c92d8fe6f34f64ab1a97c5cd830 +size 28282 diff --git a/completions/completions_00258.parquet b/completions/completions_00258.parquet new file mode 100644 index 0000000..e9b8fc4 --- /dev/null +++ b/completions/completions_00258.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:66834e7a18c1c91d705e19b1adc0814fc5012ded7d9794f3a2401980b35a8ed3 +size 27091 diff --git a/completions/completions_00259.parquet b/completions/completions_00259.parquet new file mode 100644 index 0000000..fa137fa --- /dev/null +++ b/completions/completions_00259.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1f15a612b6f2cfbd24f76907e2ab710d12b80dcdf54583ccff84a1a3fbb5225e +size 27350 diff --git a/completions/completions_00260.parquet b/completions/completions_00260.parquet new file mode 100644 index 0000000..935d825 --- /dev/null +++ b/completions/completions_00260.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:268a43fbbe9db1488b1e6621254d04d2969d3d540064af7e6cc5b72b23fa3dfa +size 26152 diff --git a/completions/completions_00261.parquet b/completions/completions_00261.parquet new file mode 100644 index 0000000..c029d16 --- /dev/null +++ b/completions/completions_00261.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f52fd16478bd2ea3babd55719f3009932fb8af80e85e50ba56a6a6da940bc671 +size 28090 diff --git a/completions/completions_00262.parquet b/completions/completions_00262.parquet new file mode 100644 index 0000000..2479c53 --- /dev/null +++ b/completions/completions_00262.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d669967c5ff98093e0165ff8503959338a15cb571a187a19a5170cc375d4e84 +size 28037 diff --git a/completions/completions_00263.parquet b/completions/completions_00263.parquet new file mode 100644 index 0000000..9e4b86a --- /dev/null +++ b/completions/completions_00263.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b116fd9e205ceb297a4b0782b2e9d008bd5507b00cd21138a7ded1c35312511f +size 26263 diff --git a/completions/completions_00264.parquet b/completions/completions_00264.parquet new file mode 100644 index 0000000..78d2207 --- /dev/null +++ b/completions/completions_00264.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:add59acaed0f28333e7f78f80bfab3c5bb8dd43b6728484f89f7a42115b58de0 +size 27449 diff --git a/completions/completions_00265.parquet b/completions/completions_00265.parquet new file mode 100644 index 0000000..cfb0a27 --- /dev/null +++ b/completions/completions_00265.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dfd3cdd4f1b97dc71788d45182e724eb9ff807a973667b7e864b8839ef668093 +size 32114 diff --git a/completions/completions_00266.parquet b/completions/completions_00266.parquet new file mode 100644 index 0000000..4e0026a --- /dev/null +++ b/completions/completions_00266.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cdc79132cb678eba65eb87a9c348d3d2c080fdbf6394bac78dcdd2e7270a3237 +size 28074 diff --git a/completions/completions_00267.parquet b/completions/completions_00267.parquet new file mode 100644 index 0000000..999e646 --- /dev/null +++ b/completions/completions_00267.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b6322c399d4d13417b7e64544014c7d1d004e537e560ba1bf429718f65cf1a0e +size 28175 diff --git a/completions/completions_00268.parquet b/completions/completions_00268.parquet new file mode 100644 index 0000000..5f41acc --- /dev/null +++ b/completions/completions_00268.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04615dbd3ce9c7b8546f5151260400103152561e0ab553c2b198b9a1181b4466 +size 23400 diff --git a/completions/completions_00269.parquet b/completions/completions_00269.parquet new file mode 100644 index 0000000..fdb284c --- /dev/null +++ b/completions/completions_00269.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:22845fda6de66fecbe85d14e62762ab96ee58c80a1d65d25eaff8fe7676039f8 +size 30657 diff --git a/completions/completions_00270.parquet b/completions/completions_00270.parquet new file mode 100644 index 0000000..c6567c4 --- /dev/null +++ b/completions/completions_00270.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:572837ec18e4146393e62c2b7f643f0e96e63fc77b54689e725faf5aeec50623 +size 33175 diff --git a/completions/completions_00271.parquet b/completions/completions_00271.parquet new file mode 100644 index 0000000..ad3d6cc --- /dev/null +++ b/completions/completions_00271.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea8a2fee7cba10dc0209ce0a97720c9567fcb849fc763b39b896a3925381c2e3 +size 28615 diff --git a/completions/completions_00272.parquet b/completions/completions_00272.parquet new file mode 100644 index 0000000..2592c3c --- /dev/null +++ b/completions/completions_00272.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1c9bd60d1583aa3b3b56e676b1c93cd5384d26c01a2862603e33aea09ff5054 +size 27309 diff --git a/completions/completions_00273.parquet b/completions/completions_00273.parquet new file mode 100644 index 0000000..f7ebfd7 --- /dev/null +++ b/completions/completions_00273.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:df56cb827381558ed9b6f37b936c3508c1553cebcdb3d3601f19637452e987c3 +size 28878 diff --git a/completions/completions_00274.parquet b/completions/completions_00274.parquet new file mode 100644 index 0000000..9eb1dda --- /dev/null +++ b/completions/completions_00274.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1e61cff7af14922d7e429880e5520465dde4d01d52961fa7b7f123ef7a5724bd +size 26304 diff --git a/completions/completions_00275.parquet b/completions/completions_00275.parquet new file mode 100644 index 0000000..5e1870e --- /dev/null +++ b/completions/completions_00275.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee0df22468f0230b5381856b29e0df59a4ba32fcb995eaa826ac5615685649f7 +size 33816 diff --git a/completions/completions_00276.parquet b/completions/completions_00276.parquet new file mode 100644 index 0000000..84bdfcd --- /dev/null +++ b/completions/completions_00276.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:647093b8619980f8c364c47e30bed14ee57898e9d748815599aa3b6eb8bfd8bf +size 30149 diff --git a/completions/completions_00277.parquet b/completions/completions_00277.parquet new file mode 100644 index 0000000..dacbefb --- /dev/null +++ b/completions/completions_00277.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ae0530129a70918390064a90be78a411cc5d55c3f3c4ae88951dac92a6abc3cc +size 23075 diff --git a/completions/completions_00278.parquet b/completions/completions_00278.parquet new file mode 100644 index 0000000..4c93c53 --- /dev/null +++ b/completions/completions_00278.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ec9bcb425a0cd21535da3b03887e5d16c1df477dfcf00996369b2a8d0cdb52a +size 28538 diff --git a/completions/completions_00279.parquet b/completions/completions_00279.parquet new file mode 100644 index 0000000..fc7cd5a --- /dev/null +++ b/completions/completions_00279.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:82acdf4ccd754958ca25d0065ca95614a7387c283eff8d7a56e3b16ae378551f +size 27073 diff --git a/completions/completions_00280.parquet b/completions/completions_00280.parquet new file mode 100644 index 0000000..7236469 --- /dev/null +++ b/completions/completions_00280.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76ec53b5a6c0794f39c3bb6cae6459df2a2e4aa4ffbf50aebf47abd0434a8d82 +size 28506 diff --git a/completions/completions_00281.parquet b/completions/completions_00281.parquet new file mode 100644 index 0000000..4e67b06 --- /dev/null +++ b/completions/completions_00281.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6404020c92762b69d446a0573acd70e7e152d449b61e1f7036fc199ad3f4d26 +size 32298 diff --git a/completions/completions_00282.parquet b/completions/completions_00282.parquet new file mode 100644 index 0000000..10e552e --- /dev/null +++ b/completions/completions_00282.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c2f05d0abe50d67f0401e20a14c9cd6b96c1982bf23e885a2aaf609aec92022a +size 27672 diff --git a/completions/completions_00283.parquet b/completions/completions_00283.parquet new file mode 100644 index 0000000..28d9680 --- /dev/null +++ b/completions/completions_00283.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ff55b147702f40306072038ff5c821af3af582024b6042c49af45dfdb6ab731 +size 27315 diff --git a/completions/completions_00284.parquet b/completions/completions_00284.parquet new file mode 100644 index 0000000..8ce3955 --- /dev/null +++ b/completions/completions_00284.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:81cd2a1325c8599c0f7bd09e006ed9bf51328e5cdc4e00eb4376f7eed5548302 +size 29408 diff --git a/completions/completions_00285.parquet b/completions/completions_00285.parquet new file mode 100644 index 0000000..29a80dd --- /dev/null +++ b/completions/completions_00285.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d3acd8439e7ba2b0471c61e3b5272b507f33ac12c9b3e47361456fe3bf85b69f +size 28205 diff --git a/completions/completions_00286.parquet b/completions/completions_00286.parquet new file mode 100644 index 0000000..b520073 --- /dev/null +++ b/completions/completions_00286.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a57a12ea0bd90565b4fcdb1d95b6b5680d66bdb6aff46b467c94a20ed4d8b52 +size 28053 diff --git a/completions/completions_00287.parquet b/completions/completions_00287.parquet new file mode 100644 index 0000000..f3d136c --- /dev/null +++ b/completions/completions_00287.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:beebb60c74dc560dd9aa8d46c7958d04715cb563971701b7eb9176d994f1e615 +size 32842 diff --git a/completions/completions_00288.parquet b/completions/completions_00288.parquet new file mode 100644 index 0000000..f5177b0 --- /dev/null +++ b/completions/completions_00288.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:26d472f40cf2f143c2fcde80ec3f23267f9777a7fb07fce2a753176a43121247 +size 31918 diff --git a/completions/completions_00289.parquet b/completions/completions_00289.parquet new file mode 100644 index 0000000..e9497f1 --- /dev/null +++ b/completions/completions_00289.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3967e89e4538784634a0240d112475f4963636b6bd156f4c8c4a16e77d23b216 +size 28865 diff --git a/completions/completions_00290.parquet b/completions/completions_00290.parquet new file mode 100644 index 0000000..397b643 --- /dev/null +++ b/completions/completions_00290.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c727bb7a311ba9a2368aeaf92fb0dac569e43e335ecd3af3230f393c5cf5651 +size 23184 diff --git a/completions/completions_00291.parquet b/completions/completions_00291.parquet new file mode 100644 index 0000000..1b0f70f --- /dev/null +++ b/completions/completions_00291.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:737abb9d96b7453e32ff6d760c978ea84385db316eb4b200baac3fccd15586c5 +size 28067 diff --git a/completions/completions_00292.parquet b/completions/completions_00292.parquet new file mode 100644 index 0000000..2567d0f --- /dev/null +++ b/completions/completions_00292.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:823a660d273a3595530eb270acc102ff8ad0f14bfd1a5d889813dbdb633feee3 +size 27799 diff --git a/completions/completions_00293.parquet b/completions/completions_00293.parquet new file mode 100644 index 0000000..3ba58df --- /dev/null +++ b/completions/completions_00293.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:24e0a4e94bee4fd9e13a8d6c1f2cfe33078eaa1213b5a12e6bdd982bee64ef4d +size 27379 diff --git a/completions/completions_00294.parquet b/completions/completions_00294.parquet new file mode 100644 index 0000000..3313f53 --- /dev/null +++ b/completions/completions_00294.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c371c0846f83e063b38086d20d17bb2393e549f7c4fec997cdd749ec6e26727e +size 28455 diff --git a/completions/completions_00295.parquet b/completions/completions_00295.parquet new file mode 100644 index 0000000..501f45e --- /dev/null +++ b/completions/completions_00295.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e03099d755a9d59605fe33a93511957c24512185713616f6198b8fad4983077 +size 31109 diff --git a/completions/completions_00296.parquet b/completions/completions_00296.parquet new file mode 100644 index 0000000..3aa0548 --- /dev/null +++ b/completions/completions_00296.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d169daaf9bef2768e88c4df4a9dfc6ddf9df1a7a0709e13dcc267145ed2982bb +size 32086 diff --git a/completions/completions_00297.parquet b/completions/completions_00297.parquet new file mode 100644 index 0000000..1bf0837 --- /dev/null +++ b/completions/completions_00297.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:85debefb797c6b232ee63f289b613fde679cff44fc60273a190445317efeb9dd +size 27385 diff --git a/completions/completions_00298.parquet b/completions/completions_00298.parquet new file mode 100644 index 0000000..a81bf8a --- /dev/null +++ b/completions/completions_00298.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d48457b9be71235e3059d25f2b9e3828330713f93aaf9bd0142da0fc99d1b8b +size 26784 diff --git a/completions/completions_00299.parquet b/completions/completions_00299.parquet new file mode 100644 index 0000000..0609dc2 --- /dev/null +++ b/completions/completions_00299.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ff531a20879aeb9ac0e468886cde1e5f4cba430d70dc29f66740ca39f654b1d4 +size 27519 diff --git a/completions/completions_00300.parquet b/completions/completions_00300.parquet new file mode 100644 index 0000000..0d7afdf --- /dev/null +++ b/completions/completions_00300.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:38b1c5a8d4be8ddb3de41a2bc483cafd35bdaee354e16ec38d4f1ef4534515ab +size 32454 diff --git a/completions/completions_00301.parquet b/completions/completions_00301.parquet new file mode 100644 index 0000000..1805315 --- /dev/null +++ b/completions/completions_00301.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:57d7e4b2d2425890f669d531f8cde1f9978c5263f42b4a9fefc39e3b1b421609 +size 32088 diff --git a/completions/completions_00302.parquet b/completions/completions_00302.parquet new file mode 100644 index 0000000..4ba293a --- /dev/null +++ b/completions/completions_00302.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5aeabff3159d2cd224ab02554c45e9ea52f4d7358fcdd3c7cd19f646ae3bf0b6 +size 27861 diff --git a/completions/completions_00303.parquet b/completions/completions_00303.parquet new file mode 100644 index 0000000..a0b8194 --- /dev/null +++ b/completions/completions_00303.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e5f2a24860276502f03aacbba8febba3008bb41289ad9b909124d6adeca4005a +size 31594 diff --git a/completions/completions_00304.parquet b/completions/completions_00304.parquet new file mode 100644 index 0000000..58278e0 --- /dev/null +++ b/completions/completions_00304.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a9a0e41383cd28daec826d27f2efae40b8dbd4e78ee0970fa563535ac06e55a +size 27524 diff --git a/completions/completions_00305.parquet b/completions/completions_00305.parquet new file mode 100644 index 0000000..1e9d0fa --- /dev/null +++ b/completions/completions_00305.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b113419c804cf2723f4056ee15a52035a507c335fe5d0420ef8b24019300e00 +size 33827 diff --git a/completions/completions_00306.parquet b/completions/completions_00306.parquet new file mode 100644 index 0000000..f071a1c --- /dev/null +++ b/completions/completions_00306.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3276673034787cab86aa0dc1677376e15ce4980b76cb201bc41602ca7ede0c25 +size 31172 diff --git a/completions/completions_00307.parquet b/completions/completions_00307.parquet new file mode 100644 index 0000000..8355c90 --- /dev/null +++ b/completions/completions_00307.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0dd28fe931d1b6ff6b2f96c38f2d25a83a0d523ec49d9afcf5aab89253c2c2ba +size 28342 diff --git a/completions/completions_00308.parquet b/completions/completions_00308.parquet new file mode 100644 index 0000000..48c9769 --- /dev/null +++ b/completions/completions_00308.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:13112215f53dc8db948e5e7b5f8d05940bb2d3f0099a70ff69960869a5eea026 +size 34372 diff --git a/completions/completions_00309.parquet b/completions/completions_00309.parquet new file mode 100644 index 0000000..35dd6cc --- /dev/null +++ b/completions/completions_00309.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8dd02efe20d23dfbf35daf8422e5fc74f654ecb5e2cf9a2efa5e104a352a79d4 +size 31672 diff --git a/completions/completions_00310.parquet b/completions/completions_00310.parquet new file mode 100644 index 0000000..759fc8b --- /dev/null +++ b/completions/completions_00310.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b091e6b22db491a8e3300698b922693d84f02bce2ddf14fc4492abf21f534be +size 26591 diff --git a/completions/completions_00311.parquet b/completions/completions_00311.parquet new file mode 100644 index 0000000..851e261 --- /dev/null +++ b/completions/completions_00311.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62cd48559fd9c70028f8437bb38f3e1934be5c3d3c546e0fd39c89a6b2cb37ad +size 31956 diff --git a/completions/completions_00312.parquet b/completions/completions_00312.parquet new file mode 100644 index 0000000..a0d96da --- /dev/null +++ b/completions/completions_00312.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:153cbc6ad2dc7873628ab118bada13e1b8d9a2f8ef1d0f5b7e1a63f7692d192b +size 27112 diff --git a/completions/completions_00313.parquet b/completions/completions_00313.parquet new file mode 100644 index 0000000..385c130 --- /dev/null +++ b/completions/completions_00313.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1bf2b7cbd6914537827f2dd529804cb93fdd43e214db88b1919869fd934256b0 +size 28051 diff --git a/completions/completions_00314.parquet b/completions/completions_00314.parquet new file mode 100644 index 0000000..a6b01bd --- /dev/null +++ b/completions/completions_00314.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:54783ba5b8dfc84f1d0e191856350f686c0cd1eb808ba32e20a053631d5da1c4 +size 31388 diff --git a/completions/completions_00315.parquet b/completions/completions_00315.parquet new file mode 100644 index 0000000..60578d5 --- /dev/null +++ b/completions/completions_00315.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:68849749b46f51ebf6c23c86fd8d63467b9af9714c220fcf178c6ee6473f22a7 +size 27205 diff --git a/completions/completions_00316.parquet b/completions/completions_00316.parquet new file mode 100644 index 0000000..f9582dd --- /dev/null +++ b/completions/completions_00316.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a686b6f642675674fb14e23ecafe9e0bdde969f0f3d45c6fca8ab78761e9c0ac +size 31363 diff --git a/completions/completions_00317.parquet b/completions/completions_00317.parquet new file mode 100644 index 0000000..6be0d28 --- /dev/null +++ b/completions/completions_00317.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:30f019151e4d2b24ad7e2696da490eddd670a1ef301e6b51de572dd87edfbd19 +size 27250 diff --git a/completions/completions_00318.parquet b/completions/completions_00318.parquet new file mode 100644 index 0000000..7c8ba03 --- /dev/null +++ b/completions/completions_00318.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a78bfabbe769e31e49a3d6c029ba0c867fbd55acfb1e3eb48bb4bbfe2a37dc7d +size 23304 diff --git a/completions/completions_00319.parquet b/completions/completions_00319.parquet new file mode 100644 index 0000000..e0d037c --- /dev/null +++ b/completions/completions_00319.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:065ea4bd5e29ddda2038a2e0e10c7510709b35b928055e2e34734d3690f151a2 +size 27555 diff --git a/completions/completions_00320.parquet b/completions/completions_00320.parquet new file mode 100644 index 0000000..e1eff9d --- /dev/null +++ b/completions/completions_00320.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5f61c5dead72fd64d2ddf3d05555ff25231535bc59ab564c93944b45870e0af7 +size 31387 diff --git a/completions/completions_00321.parquet b/completions/completions_00321.parquet new file mode 100644 index 0000000..94c7805 --- /dev/null +++ b/completions/completions_00321.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:07663a6d54d094ab8f2585842b8b5cd4405d75985baa6c3899d637fbced0e04a +size 30579 diff --git a/completions/completions_00322.parquet b/completions/completions_00322.parquet new file mode 100644 index 0000000..9ee8bbe --- /dev/null +++ b/completions/completions_00322.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4d774ee46d0c36626d8d2613e623015ab014a44a448bb5d8b1b87ffbc3bc4dfe +size 32462 diff --git a/completions/completions_00323.parquet b/completions/completions_00323.parquet new file mode 100644 index 0000000..01a5a7c --- /dev/null +++ b/completions/completions_00323.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5c8bcf90494b147ab211dce08924b30e13552da49fcf29eb05b353d885e1df14 +size 27672 diff --git a/completions/completions_00324.parquet b/completions/completions_00324.parquet new file mode 100644 index 0000000..01cd1cf --- /dev/null +++ b/completions/completions_00324.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ee3dd3b60bf4dbb9727c399efeedead61099c2cce63f27c437d3deecef42261 +size 29005 diff --git a/completions/completions_00325.parquet b/completions/completions_00325.parquet new file mode 100644 index 0000000..ed53a38 --- /dev/null +++ b/completions/completions_00325.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2a9e3ef2c70fd5e911959ef39459f31dcfe2a102f440688218fc221b612828f2 +size 28662 diff --git a/completions/completions_00326.parquet b/completions/completions_00326.parquet new file mode 100644 index 0000000..b5f58a8 --- /dev/null +++ b/completions/completions_00326.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92937bde68dfa74555d59ecefffc8149ed1d11f22dc041dd3ab2022680b0c1cc +size 26976 diff --git a/completions/completions_00327.parquet b/completions/completions_00327.parquet new file mode 100644 index 0000000..8fd0e88 --- /dev/null +++ b/completions/completions_00327.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee21586ae9b507a871a159daf8504a6df0303ffe234bca6ef1eb5e62ba059294 +size 22613 diff --git a/completions/completions_00328.parquet b/completions/completions_00328.parquet new file mode 100644 index 0000000..fcb57c2 --- /dev/null +++ b/completions/completions_00328.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b872fbb09f76a3ab2ac2bb2b90e686a27df5039853db677f6cb00d8a1c111a19 +size 22476 diff --git a/completions/completions_00329.parquet b/completions/completions_00329.parquet new file mode 100644 index 0000000..189aa2d --- /dev/null +++ b/completions/completions_00329.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1bedfd295535ec726be8bc4f6a063526184923ae88833544ff7c58531b6b95a +size 27797 diff --git a/completions/completions_00330.parquet b/completions/completions_00330.parquet new file mode 100644 index 0000000..9dcef92 --- /dev/null +++ b/completions/completions_00330.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f171ca421f1be49f2a9c0d17fb6c02e9e1c1f422678b8f5fe7f16b94f68fc54b +size 27135 diff --git a/completions/completions_00331.parquet b/completions/completions_00331.parquet new file mode 100644 index 0000000..5c3bf76 --- /dev/null +++ b/completions/completions_00331.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cdd9bbb552fa4ff4ec66a51e971773effdb23c8ab9622506a47f020bd92b1dd2 +size 22711 diff --git a/completions/completions_00332.parquet b/completions/completions_00332.parquet new file mode 100644 index 0000000..9f936db --- /dev/null +++ b/completions/completions_00332.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fdeb265b2b0663227a85b247fc33fb8e605cf45be938004bce878b94fc35353 +size 31292 diff --git a/completions/completions_00333.parquet b/completions/completions_00333.parquet new file mode 100644 index 0000000..333c166 --- /dev/null +++ b/completions/completions_00333.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ae555293c60fcee2e103131c9e18bf09e09b5f3f127d46eefc9eea8882db5d8 +size 27476 diff --git a/completions/completions_00334.parquet b/completions/completions_00334.parquet new file mode 100644 index 0000000..093a99f --- /dev/null +++ b/completions/completions_00334.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca5335b487aba07685fc4a3d44736638515e6c3f2abf8137bdc4fd6899bc2fdc +size 27843 diff --git a/completions/completions_00335.parquet b/completions/completions_00335.parquet new file mode 100644 index 0000000..13a1ff2 --- /dev/null +++ b/completions/completions_00335.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b749433d4b44428d1aefe4298766716bd044bd818c612944d193cfd8fb26a10d +size 26038 diff --git a/completions/completions_00336.parquet b/completions/completions_00336.parquet new file mode 100644 index 0000000..91efae7 --- /dev/null +++ b/completions/completions_00336.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbd9c5a60866d2bf9772d3f85df9bc5a1eeb7c48dff4e74c93c75bb0a57a211a +size 22767 diff --git a/completions/completions_00337.parquet b/completions/completions_00337.parquet new file mode 100644 index 0000000..7c574e6 --- /dev/null +++ b/completions/completions_00337.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a0b892353f16aace3d726ece3e0cd75cfd24f61939bc0e445ba7494f095fc87d +size 31079 diff --git a/completions/completions_00338.parquet b/completions/completions_00338.parquet new file mode 100644 index 0000000..d713a9a --- /dev/null +++ b/completions/completions_00338.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7212760ac527a5fb528e43fde33387d8950734b0475db28f8a5651ec80bb1ee6 +size 27634 diff --git a/completions/completions_00339.parquet b/completions/completions_00339.parquet new file mode 100644 index 0000000..40ef318 --- /dev/null +++ b/completions/completions_00339.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bd066bedaafcf321befdb72859b5e7ec51f1774750dc37d31dbeac39e550e05a +size 32573 diff --git a/completions/completions_00340.parquet b/completions/completions_00340.parquet new file mode 100644 index 0000000..766ca35 --- /dev/null +++ b/completions/completions_00340.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:871b8c7bbc880c7e2323d265430014459c045d9dd46aadceb149ac9f9fda271d +size 22834 diff --git a/completions/completions_00341.parquet b/completions/completions_00341.parquet new file mode 100644 index 0000000..8dc13e6 --- /dev/null +++ b/completions/completions_00341.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc510a93cc04d125c373cece867c179f5604e2c2abc4dd580f9978ff78c79961 +size 27567 diff --git a/completions/completions_00342.parquet b/completions/completions_00342.parquet new file mode 100644 index 0000000..d52b323 --- /dev/null +++ b/completions/completions_00342.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2b835ac9997c1074403f4cf9b0a6a1b4623a1c29d8142442eada3d0a11894ccc +size 28776 diff --git a/completions/completions_00343.parquet b/completions/completions_00343.parquet new file mode 100644 index 0000000..d09a966 --- /dev/null +++ b/completions/completions_00343.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d850b322aed28f679d43c6381baa27dcd55ed67ee7479991f18fd5484b802d7c +size 32729 diff --git a/completions/completions_00344.parquet b/completions/completions_00344.parquet new file mode 100644 index 0000000..70e71a4 --- /dev/null +++ b/completions/completions_00344.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b1b08202de54b7044db14e97acd883e2a1a6b45d507f2a2649015271825b8b3b +size 22908 diff --git a/completions/completions_00345.parquet b/completions/completions_00345.parquet new file mode 100644 index 0000000..b3ccfec --- /dev/null +++ b/completions/completions_00345.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6af6da36513b56d87546a58e8c29cd0dc343fe889ed2ad911b10c530f3740735 +size 22823 diff --git a/completions/completions_00346.parquet b/completions/completions_00346.parquet new file mode 100644 index 0000000..d006041 --- /dev/null +++ b/completions/completions_00346.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abb31f7798a3fb5940477a64d2ae02b2f7179746765900a25f3ea9a3934eb51f +size 31687 diff --git a/completions/completions_00347.parquet b/completions/completions_00347.parquet new file mode 100644 index 0000000..293bf70 --- /dev/null +++ b/completions/completions_00347.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d6ffa27881215ff7c7d75fc247207a2c57e9ebdc13890aec40a337a9194fe7a +size 31733 diff --git a/completions/completions_00348.parquet b/completions/completions_00348.parquet new file mode 100644 index 0000000..24ae1c6 --- /dev/null +++ b/completions/completions_00348.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53bcc4bbcd43e7e8d633fa6904c5648961ebef1fd838c5f37eee1132a9c7ff1d +size 27221 diff --git a/completions/completions_00349.parquet b/completions/completions_00349.parquet new file mode 100644 index 0000000..58c634d --- /dev/null +++ b/completions/completions_00349.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:423b34bb798353b9626270e5a11e508c1ccede234deb5fa0b3b43929c348488e +size 25794 diff --git a/completions/completions_00350.parquet b/completions/completions_00350.parquet new file mode 100644 index 0000000..503c1f9 --- /dev/null +++ b/completions/completions_00350.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b1ef2ac566b7813acf48cfb8285bb3da7215c7d2a8ffe4c4eccb10659be59f30 +size 27989 diff --git a/completions/completions_00351.parquet b/completions/completions_00351.parquet new file mode 100644 index 0000000..09854cd --- /dev/null +++ b/completions/completions_00351.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:25ea8f847536972dff8ac4f10791257c6817373fe6632bccb5b54e2e756ee3d1 +size 28106 diff --git a/completions/completions_00352.parquet b/completions/completions_00352.parquet new file mode 100644 index 0000000..a1e8ce3 --- /dev/null +++ b/completions/completions_00352.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a5b0fd505ca5b1def56d8e0179604c512d81b3c656c14ed52df4b7546e623b82 +size 32122 diff --git a/completions/completions_00353.parquet b/completions/completions_00353.parquet new file mode 100644 index 0000000..d3b83a1 --- /dev/null +++ b/completions/completions_00353.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5dde1b12eec627ecfed18b839c9b8505ede71486df284420918694a55d603b2d +size 28467 diff --git a/completions/completions_00354.parquet b/completions/completions_00354.parquet new file mode 100644 index 0000000..230076c --- /dev/null +++ b/completions/completions_00354.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:44b7296f3b65b0f22b02384b6b35d40ce2d988b64dbe204d636442ba6d326f9f +size 22852 diff --git a/completions/completions_00355.parquet b/completions/completions_00355.parquet new file mode 100644 index 0000000..6f58791 --- /dev/null +++ b/completions/completions_00355.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bee00aaf30cb7e1fdba44c97bfe67799e1e059588fdae34968741d77407bd3f0 +size 22981 diff --git a/completions/completions_00356.parquet b/completions/completions_00356.parquet new file mode 100644 index 0000000..e039f25 --- /dev/null +++ b/completions/completions_00356.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a47e7d8da1f00b8f0e66f78e8f5ee7679ed9aa9f15d82bfe702b7989127cac07 +size 26714 diff --git a/completions/completions_00357.parquet b/completions/completions_00357.parquet new file mode 100644 index 0000000..3a32055 --- /dev/null +++ b/completions/completions_00357.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eff9160d43790dc82e70f0e9e8eb5b5d39d1f6f11328a9f6c80e4b28c8a3f23c +size 31716 diff --git a/completions/completions_00358.parquet b/completions/completions_00358.parquet new file mode 100644 index 0000000..685e425 --- /dev/null +++ b/completions/completions_00358.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f75d4cb95ba230fc4580f5e1b578aafad8a087e6c21b6c0cf26cfc190f00664b +size 28515 diff --git a/completions/completions_00359.parquet b/completions/completions_00359.parquet new file mode 100644 index 0000000..03af9d2 --- /dev/null +++ b/completions/completions_00359.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:374b4e9208bbae704b6a81f14905744857eb412e30ca04c08014ad9d9152ac4e +size 29120 diff --git a/completions/completions_00360.parquet b/completions/completions_00360.parquet new file mode 100644 index 0000000..4f0eb70 --- /dev/null +++ b/completions/completions_00360.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:413cd14e5b85f2c7dc620244a4675742e414824056a57184f1b09d83001f0268 +size 30021 diff --git a/completions/completions_00361.parquet b/completions/completions_00361.parquet new file mode 100644 index 0000000..c2b28a0 --- /dev/null +++ b/completions/completions_00361.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1ca92a2f720961c8a0bf339973fbab8b0d98d969bc32ea276a207b09839b2a52 +size 27601 diff --git a/completions/completions_00362.parquet b/completions/completions_00362.parquet new file mode 100644 index 0000000..5135216 --- /dev/null +++ b/completions/completions_00362.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:30408053994fcd85e8f31dcc093dc4cca783a0d58a6c0275a1b0286c3d9a17db +size 34570 diff --git a/completions/completions_00363.parquet b/completions/completions_00363.parquet new file mode 100644 index 0000000..2e82299 --- /dev/null +++ b/completions/completions_00363.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:674c090e34c0f9bd6c2b6941a0323c940324635b59946c08ac6a5662fe26be0d +size 22672 diff --git a/completions/completions_00364.parquet b/completions/completions_00364.parquet new file mode 100644 index 0000000..0edc13e --- /dev/null +++ b/completions/completions_00364.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5c82126ebb78094809443dc3d2cb34fdbc2cc3a9ed111dcaea64c83650d0c76f +size 33612 diff --git a/completions/completions_00365.parquet b/completions/completions_00365.parquet new file mode 100644 index 0000000..cc66f48 --- /dev/null +++ b/completions/completions_00365.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:64c63014d4808018ae7f5c4fb2ddadd44f57dfd305ffd4d7aba3c4385b89678b +size 33636 diff --git a/completions/completions_00366.parquet b/completions/completions_00366.parquet new file mode 100644 index 0000000..ea4d3d8 --- /dev/null +++ b/completions/completions_00366.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:acf8caebbcf79e20e15ebbdaca6825bbdf3ad08387c6841ffa5c140408707f13 +size 28059 diff --git a/completions/completions_00367.parquet b/completions/completions_00367.parquet new file mode 100644 index 0000000..fec7990 --- /dev/null +++ b/completions/completions_00367.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e9d66d242f07eba93504e6e5ae3a7b4040bc839a488fe5f5e74e4397c723739e +size 27143 diff --git a/completions/completions_00368.parquet b/completions/completions_00368.parquet new file mode 100644 index 0000000..1f32840 --- /dev/null +++ b/completions/completions_00368.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f104fa809ffefad35f68c857c66e387bdbab6a9b51c337cc18cbbea10d2fce5 +size 32966 diff --git a/completions/completions_00369.parquet b/completions/completions_00369.parquet new file mode 100644 index 0000000..40805a5 --- /dev/null +++ b/completions/completions_00369.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b9d5527b446d9fdf316e45d5a95c44ab1476948cf6c65335ebad920e3c4c619d +size 23272 diff --git a/completions/completions_00370.parquet b/completions/completions_00370.parquet new file mode 100644 index 0000000..abbc9ad --- /dev/null +++ b/completions/completions_00370.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:36fbc7d9a4f58f2dd2f0ee05e9d59ee8cbf066e4dc2608392415be0b86d2701e +size 32875 diff --git a/completions/completions_00371.parquet b/completions/completions_00371.parquet new file mode 100644 index 0000000..1f02fbd --- /dev/null +++ b/completions/completions_00371.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5704e4e3ed8973d695de07d32e89eecaa72c9b8e7ce5a69a579fe2c884c129f0 +size 27981 diff --git a/completions/completions_00372.parquet b/completions/completions_00372.parquet new file mode 100644 index 0000000..cafdf03 --- /dev/null +++ b/completions/completions_00372.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0111a932733a6e02c70fcabca5d2765c88524c338a40c5e9da19ead2fcc51054 +size 31240 diff --git a/completions/completions_00373.parquet b/completions/completions_00373.parquet new file mode 100644 index 0000000..eda3abe --- /dev/null +++ b/completions/completions_00373.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e9fdc132a72eb5a0936aa6866fddb0bf0ec58372f1ec6cffe08a9f0e0f26cd8 +size 27629 diff --git a/completions/completions_00374.parquet b/completions/completions_00374.parquet new file mode 100644 index 0000000..367bd71 --- /dev/null +++ b/completions/completions_00374.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5329b86b902602ff910a8585c93776e948a2effaf55b161b393214111b9bf6ef +size 28942 diff --git a/completions/completions_00375.parquet b/completions/completions_00375.parquet new file mode 100644 index 0000000..9b6517e --- /dev/null +++ b/completions/completions_00375.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4205d23d8e2ab4f71feec3460a4c091a53ce493716ae678be7e6ebb0eaa4b4d3 +size 23186 diff --git a/completions/completions_00376.parquet b/completions/completions_00376.parquet new file mode 100644 index 0000000..24d15a0 --- /dev/null +++ b/completions/completions_00376.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8eb7158c09a8d7d9ac07126ff62522cb3e45345b2f74846bc5a200652ccb0300 +size 32689 diff --git a/completions/completions_00377.parquet b/completions/completions_00377.parquet new file mode 100644 index 0000000..7ffe545 --- /dev/null +++ b/completions/completions_00377.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0562f2b052589fa321780b426bd1e30151f5532f05cbc54dc84230e81fa7a736 +size 23251 diff --git a/completions/completions_00378.parquet b/completions/completions_00378.parquet new file mode 100644 index 0000000..eee8819 --- /dev/null +++ b/completions/completions_00378.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:100d150872b39cc4476c4a98c484c4039aa486d4aa5d8b6aa0e14788b85a3665 +size 27517 diff --git a/completions/completions_00379.parquet b/completions/completions_00379.parquet new file mode 100644 index 0000000..3e2d82b --- /dev/null +++ b/completions/completions_00379.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:34ee949f9fea20051786b5f4c284c7cc3dc084d926deb36117e31a1b3e7ad353 +size 27465 diff --git a/completions/completions_00380.parquet b/completions/completions_00380.parquet new file mode 100644 index 0000000..c474dcf --- /dev/null +++ b/completions/completions_00380.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f2e5265bafe96f7b4ed988c4ca2d4b25a502f2bd382995b4317723d65622daf0 +size 28910 diff --git a/completions/completions_00381.parquet b/completions/completions_00381.parquet new file mode 100644 index 0000000..4277d69 --- /dev/null +++ b/completions/completions_00381.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6472c03e08250d09b18a2ea0eb3cb7317a39d96e6be3aea090571654fa1f4dfa +size 28432 diff --git a/completions/completions_00382.parquet b/completions/completions_00382.parquet new file mode 100644 index 0000000..81834ea --- /dev/null +++ b/completions/completions_00382.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cc9cfbe0dbec5300372572582f04e0c2959b6b00d7b4fa9af606447df2bdce36 +size 32237 diff --git a/completions/completions_00383.parquet b/completions/completions_00383.parquet new file mode 100644 index 0000000..fd176f7 --- /dev/null +++ b/completions/completions_00383.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:641835aea9374e9d018adf783b5ddb3ffaff6cb250b022425a85a8e08ef2eb87 +size 23163 diff --git a/completions/completions_00384.parquet b/completions/completions_00384.parquet new file mode 100644 index 0000000..12e22e2 --- /dev/null +++ b/completions/completions_00384.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7107cd71a05ce1e484bd2e76a2c45f0f6516cb8c037809059dd81353cc6f0acc +size 33982 diff --git a/completions/completions_00385.parquet b/completions/completions_00385.parquet new file mode 100644 index 0000000..e3135ff --- /dev/null +++ b/completions/completions_00385.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7273082ea96f8ff25f0ace5cf206cbcceb0a0172439eb5b22cbd69f49f866957 +size 22930 diff --git a/completions/completions_00386.parquet b/completions/completions_00386.parquet new file mode 100644 index 0000000..5d0e25a --- /dev/null +++ b/completions/completions_00386.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ac084f99810c3344f2823074cb7b039c35e888a41847f5ca20426721cc7c430 +size 22839 diff --git a/completions/completions_00387.parquet b/completions/completions_00387.parquet new file mode 100644 index 0000000..c1d6322 --- /dev/null +++ b/completions/completions_00387.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b914a9c5d4f211f5ac0781620ddc7ca236daaa1178388d50a9235e8a8a3a292b +size 33601 diff --git a/completions/completions_00388.parquet b/completions/completions_00388.parquet new file mode 100644 index 0000000..4f22852 --- /dev/null +++ b/completions/completions_00388.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a7969531ba38970355fc610195f996ff6e6d361fc704f850da82495fd1aae0f7 +size 31557 diff --git a/completions/completions_00389.parquet b/completions/completions_00389.parquet new file mode 100644 index 0000000..27451be --- /dev/null +++ b/completions/completions_00389.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d765cbd92227d68de09c15a1ef2e2cb044db146a136d435cd41780d8ad5b8f5 +size 26214 diff --git a/completions/completions_00390.parquet b/completions/completions_00390.parquet new file mode 100644 index 0000000..34debcb --- /dev/null +++ b/completions/completions_00390.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ae87670778810532f468e0f77b441bf471d254455b2d3b3a946ab67b7ec8d3ba +size 27367 diff --git a/completions/completions_00391.parquet b/completions/completions_00391.parquet new file mode 100644 index 0000000..31b11d6 --- /dev/null +++ b/completions/completions_00391.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:77f680d04951e98f026afcb67cef02e23580b2c5c632ae95b6dabd65d67dc8e4 +size 31117 diff --git a/completions/completions_00392.parquet b/completions/completions_00392.parquet new file mode 100644 index 0000000..4a9d47c --- /dev/null +++ b/completions/completions_00392.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ef9ef62a115fc5835824b66500a39dc3e70b7597c8c07a899f0489864b015a0 +size 32074 diff --git a/completions/completions_00393.parquet b/completions/completions_00393.parquet new file mode 100644 index 0000000..41f613e --- /dev/null +++ b/completions/completions_00393.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:64ffd5a9df99d603a18aa64b7f99f6518ce25628b411932c6737ad93402f1e91 +size 29006 diff --git a/completions/completions_00394.parquet b/completions/completions_00394.parquet new file mode 100644 index 0000000..80bd4a6 --- /dev/null +++ b/completions/completions_00394.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ccd084a4fe54e352dcf6cac1fcf354999e038a0d388d367d7d46004646356913 +size 27852 diff --git a/completions/completions_00395.parquet b/completions/completions_00395.parquet new file mode 100644 index 0000000..0a8b0c1 --- /dev/null +++ b/completions/completions_00395.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ecf5dc2df0825d49888011629f01d409e061805a5bd89df4aedd4edadefb78bf +size 26566 diff --git a/completions/completions_00396.parquet b/completions/completions_00396.parquet new file mode 100644 index 0000000..8c02b04 --- /dev/null +++ b/completions/completions_00396.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2373fa29b2d5b96790fd7988e44672e0d7520ef532d3dffb68ef37b611a62a60 +size 28209 diff --git a/completions/completions_00397.parquet b/completions/completions_00397.parquet new file mode 100644 index 0000000..f00c655 --- /dev/null +++ b/completions/completions_00397.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b4f29e80a43de05e4aa6d327bf3db84616b080d25ff1ba6480255e9728ca4ab +size 26739 diff --git a/completions/completions_00398.parquet b/completions/completions_00398.parquet new file mode 100644 index 0000000..2f4b26a --- /dev/null +++ b/completions/completions_00398.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a32393674563e5b3e8e1a36b96ecb05c972a35f5318a35b784c498735b874684 +size 28580 diff --git a/completions/completions_00399.parquet b/completions/completions_00399.parquet new file mode 100644 index 0000000..f1edd21 --- /dev/null +++ b/completions/completions_00399.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:38fdc73dbdac64466b7b9b144c9e976eac924d1723b23591dac67599282eb00a +size 27096 diff --git a/completions/completions_00400.parquet b/completions/completions_00400.parquet new file mode 100644 index 0000000..5d92932 --- /dev/null +++ b/completions/completions_00400.parquet @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de91e4772fb7444be2a08294c87eeafb52f128bc1f549208c68b2ff10f25a14e +size 33351 diff --git a/config.json b/config.json new file mode 100644 index 0000000..a59d6e2 --- /dev/null +++ b/config.json @@ -0,0 +1,63 @@ +{ + "architectures": [ + "Qwen3ForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": null, + "dtype": "float32", + "eos_token_id": 151645, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 6144, + "layer_types": [ + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention", + "full_attention" + ], + "max_position_embeddings": 40960, + "max_window_layers": 28, + "model_type": "qwen3", + "num_attention_heads": 16, + "num_hidden_layers": 28, + "num_key_value_heads": 8, + "pad_token_id": 151643, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 1000000, + "rope_type": "default" + }, + "sliding_window": null, + "tie_word_embeddings": true, + "transformers_version": "5.7.0.dev0", + "use_cache": false, + "use_sliding_window": false, + "vocab_size": 151936 +} diff --git a/evals/eval_clarify-rl-grpo-qwen3-1-7b-run7_n50_v5.json b/evals/eval_clarify-rl-grpo-qwen3-1-7b-run7_n50_v5.json new file mode 100644 index 0000000..eb6bc04 --- /dev/null +++ b/evals/eval_clarify-rl-grpo-qwen3-1-7b-run7_n50_v5.json @@ -0,0 +1,10506 @@ +{ + "summary": { + "model": "agarwalanu3103/clarify-rl-grpo-qwen3-1-7b-run7", + "mode": "api", + "scenarios_total": 50, + "elapsed_s": 119.10098934173584, + "avg_score": 0.0754010101010101, + "avg_questions": 5.48, + "format_pass_rate": 0.0, + "completion_rate": 0.2 + }, + "config": { + "mode": "api", + "model": "agarwalanu3103/clarify-rl-grpo-qwen3-1-7b-run7", + "api_base_url": "http://127.0.0.1:8000/v1", + "env_base_url": "https://agarwalanu3103-clarify-rl.hf.space", + "scenarios_file": "/tmp/clarify-rl/scenarios/eval_held_out.json", + "limit": 50 + }, + "results": [ + { + "seed": 10000, + "task_id": "easy", + "scenario_id": "seed10000_event_planning_easy", + "family": "event_planning", + "request": "Set up a celebration.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a celebration.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 4, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a office.\", \"questions_remaining\": 3, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the guest list?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme or decoration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"celebration\", \"date\": \"today.\", \"venue\": \"At a office.\", \"guest_list\": \"Up to you on that one.\", \"theme\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 4, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a office.", + "questions_remaining": 3, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the guest list?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the theme or decoration?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"celebration\", \"date\": \"today.\", \"venue\": \"At a office.\", \"guest_list\": \"Up to you on that one.\", \"theme\": \"Up to you on that one.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0 + }, + "expected_profile": { + "event_type": "birthday", + "date": "today", + "venue": "office" + }, + "critical_fields": [ + "date", + "event_type", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "celebration", + "date": "today.", + "venue": "At a office.", + "guest_list": "Up to you on that one.", + "theme": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "venue" + ] + } + } + ], + "elapsed_s": 3.3920176029205322 + }, + { + "seed": 10000, + "task_id": "medium", + "scenario_id": "seed10000_event_planning_medium", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.4061111111111111, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2222222222222222, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the guest list size?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 5 people.\", \"questions_remaining\": 3, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget for the party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Budget around $500-2000.\", \"questions_remaining\": 2, \"field_revealed\": \"budget_band\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the preferred venue for the party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a office.\", \"questions_remaining\": 1, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 5 people.\", \"budget_band\": \"Budget around $500-2000.\", \"venue\": \"At a office.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the name of the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the guest list size?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 5 people.", + "questions_remaining": 3, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the budget for the party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Budget around $500-2000.", + "questions_remaining": 2, + "field_revealed": "budget_band", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the preferred venue for the party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a office.", + "questions_remaining": 1, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 5 people.\", \"budget_band\": \"Budget around $500-2000.\", \"venue\": \"At a office.\"}" + }, + "reward": 0.4061111111111111, + "done": true, + "result": { + "type": "resolution", + "score": 0.4061111111111111, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2222222222222222, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "event_type": "birthday", + "date": "today", + "guest_count": 5, + "venue": "office", + "budget_band": "$500-2000" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "today.", + "guest_count": "About 5 people.", + "budget_band": "Budget around $500-2000.", + "venue": "At a office." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "budget_band", + "date", + "guest_count", + "venue" + ] + } + } + ], + "elapsed_s": 2.395275592803955 + }, + { + "seed": 10000, + "task_id": "hard", + "scenario_id": "seed10000_event_planning_hard", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.3784090909090909, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a office.\", \"questions_remaining\": 3, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of guests?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 100 people.\", \"questions_remaining\": 2, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What are the preferences for the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"this week.\", \"time\": \"Up to you on that one.\", \"venue\": \"At a office.\", \"guest_count\": \"About 100 people.\", \"preferences\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time of the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a office.", + "questions_remaining": 3, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of guests?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 100 people.", + "questions_remaining": 2, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What are the preferences for the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"this week.\", \"time\": \"Up to you on that one.\", \"venue\": \"At a office.\", \"guest_count\": \"About 100 people.\", \"preferences\": \"Up to you on that one.\"}" + }, + "reward": 0.3784090909090909, + "done": true, + "result": { + "type": "resolution", + "score": 0.3784090909090909, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "event_type": "birthday", + "date": "this week", + "guest_count": 100, + "venue": "office", + "budget_band": "$100-500", + "dietary_constraints": "none", + "theme": "surprise" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "this week.", + "time": "Up to you on that one.", + "venue": "At a office.", + "guest_count": "About 100 people.", + "preferences": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "guest_count", + "venue" + ] + } + } + ], + "elapsed_s": 2.52453875541687 + }, + { + "seed": 10001, + "task_id": "easy", + "scenario_id": "seed10001_support_triage_easy", + "family": "support_triage", + "request": "My order is wrong.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nMy order is wrong.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct order number?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Order none-provided.\", \"questions_remaining\": 5, \"field_revealed\": \"order_id\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct time of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct location of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct method of delivery?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "get_task_info()" + }, + { + "role": "user", + "content": "{\"request\": \"My order is wrong.\", \"task_id\": \"easy\", \"task_title\": \"Mild Ambiguity\", \"family\": \"support_triage\", \"questions_remaining\": 1, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"order_issue\", \"order_id\": \"Order none-provided.\", \"date\": \"I don't really know \\u2014 does it matter?\", \"time\": \"I don't really know \\u2014 does it matter?\", \"location\": \"I don't really know \\u2014 does it matter?\", \"delivery_method\": \"I don't really know \\u2014 does it matter?\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the correct order number?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Order none-provided.", + "questions_remaining": 5, + "field_revealed": "order_id", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the correct date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the correct time of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the correct location of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the correct method of delivery?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "My order is wrong.", + "task_id": "easy", + "task_title": "Mild Ambiguity", + "family": "support_triage", + "questions_remaining": 1, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"order_issue\", \"order_id\": \"Order none-provided.\", \"date\": \"I don't really know \\u2014 does it matter?\", \"time\": \"I don't really know \\u2014 does it matter?\", \"location\": \"I don't really know \\u2014 does it matter?\", \"delivery_method\": \"I don't really know \\u2014 does it matter?\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "order_id": "none-provided", + "item_issue": "late" + }, + "critical_fields": [ + "item_issue", + "order_id" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "order_issue", + "order_id": "Order none-provided.", + "date": "I don't really know \u2014 does it matter?", + "time": "I don't really know \u2014 does it matter?", + "location": "I don't really know \u2014 does it matter?", + "delivery_method": "I don't really know \u2014 does it matter?" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "order_id" + ] + } + } + ], + "elapsed_s": 3.516732692718506 + }, + { + "seed": 10001, + "task_id": "medium", + "scenario_id": "seed10001_support_triage_medium", + "family": "support_triage", + "request": "My order is wrong.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nMy order is wrong.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the correct date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is the event type birthday or wedding?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is the event a celebration or a funeral?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is the event a party or a gathering?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is the event a formal or informal gathering?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is the event a celebration or a mourning event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"event\": \"birthday\", \"occasion\": \"birthday\", \"formality\": \"formal\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the correct date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "Is the event type birthday or wedding?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "Is the event a celebration or a funeral?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "Is the event a party or a gathering?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "Is the event a formal or informal gathering?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "Is the event a celebration or a mourning event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"event\": \"birthday\", \"occasion\": \"birthday\", \"formality\": \"formal\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "order_id": "#9027", + "item_issue": "never-arrived", + "refund_or_replace": "unsure", + "urgency": "high" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "event": "birthday", + "occasion": "birthday", + "formality": "formal" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.6482832431793213 + }, + { + "seed": 10001, + "task_id": "hard", + "scenario_id": "seed10001_support_triage_hard", + "family": "support_triage", + "request": "There's an issue with my purchase.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nThere's an issue with my purchase.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the issue with the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the expected resolution for the issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'd prefer a refund.\", \"questions_remaining\": 1, \"field_revealed\": \"refund_or_replace\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"New York\", \"reason\": \"refund\", \"expected_resolution\": \"refund\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the reason for the issue with the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the expected resolution for the issue?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "I'd prefer a refund.", + "questions_remaining": 1, + "field_revealed": "refund_or_replace", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"New York\", \"reason\": \"refund\", \"expected_resolution\": \"refund\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.18181818181818182, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "order_id": "none-provided", + "item_issue": "late", + "refund_or_replace": "refund", + "urgency": "high", + "channel_preferred": "phone" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "purchase", + "date": "2024-12-25", + "location": "New York", + "reason": "refund", + "expected_resolution": "refund" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "refund_or_replace" + ] + } + } + ], + "elapsed_s": 2.331852674484253 + }, + { + "seed": 10002, + "task_id": "easy", + "scenario_id": "seed10002_meeting_scheduling_easy", + "family": "meeting_scheduling", + "request": "Schedule a sync.", + "final_score": 0.425, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.5, + "InfoGainRubric": 0.5, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSchedule a sync.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of attendees?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"whole team.\", \"questions_remaining\": 2, \"field_revealed\": \"participants\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Conference Room A\", \"participants\": \"whole team\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of attendees?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "whole team.", + "questions_remaining": 2, + "field_revealed": "participants", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Conference Room A\", \"participants\": \"whole team\"}" + }, + "reward": 0.425, + "done": true, + "result": { + "type": "resolution", + "score": 0.425, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.5, + "InfoGainRubric": 0.5, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "whole team", + "time": "evening" + }, + "critical_fields": [ + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "time": "18:00", + "location": "Conference Room A", + "participants": "whole team" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "participants" + ] + } + } + ], + "elapsed_s": 2.1206984519958496 + }, + { + "seed": 10002, + "task_id": "medium", + "scenario_id": "seed10002_meeting_scheduling_medium", + "family": "meeting_scheduling", + "request": "Set up a meeting.", + "final_score": 0.35, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "questions_asked": 4, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a meeting.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What time of the day is the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"morning.\", \"questions_remaining\": 4, \"field_revealed\": \"time\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Who are the participants?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"whole team.\", \"questions_remaining\": 3, \"field_revealed\": \"participants\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"meeting\", \"date\": \"this week.\", \"time\": \"morning.\", \"participants\": \"whole team.\", \"location\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the meeting?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What time of the day is the meeting?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "morning.", + "questions_remaining": 4, + "field_revealed": "time", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "Who are the participants?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "whole team.", + "questions_remaining": 3, + "field_revealed": "participants", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"meeting\", \"date\": \"this week.\", \"time\": \"morning.\", \"participants\": \"whole team.\", \"location\": \"No preference, you choose.\"}" + }, + "reward": 0.35, + "done": true, + "result": { + "type": "resolution", + "score": 0.35, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "expected_profile": { + "participants": "whole team", + "date": "this week", + "time": "morning", + "duration_minutes": 90 + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "meeting", + "date": "this week.", + "time": "morning.", + "participants": "whole team.", + "location": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 4, + "fields_revealed": [ + "date", + "participants", + "time" + ] + } + } + ], + "elapsed_s": 1.867630958557129 + }, + { + "seed": 10002, + "task_id": "hard", + "scenario_id": "seed10002_meeting_scheduling_hard", + "family": "meeting_scheduling", + "request": "Get a call on the calendar.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nGet a call on the calendar.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"event_name\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"event_name\": \"No preference, you choose.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "expected_profile": { + "participants": "leadership", + "date": "today", + "time": "evening", + "duration_minutes": 30, + "platform": "in-person" + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "today.", + "time": "No preference, you choose.", + "location": "No preference, you choose.", + "event_name": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date" + ] + } + } + ], + "elapsed_s": 2.0554332733154297 + }, + { + "seed": 10003, + "task_id": "easy", + "scenario_id": "seed10003_medical_intake_easy", + "family": "medical_intake", + "request": "I have a problem.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI have a problem.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Grand Hall A\", \"time\": \"18:00\", \"reason\": \"Celebrating 10 years of friendship\", \"name_of_event\": \"10th Birthday Party\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Grand Hall A\", \"time\": \"18:00\", \"reason\": \"Celebrating 10 years of friendship\", \"name_of_event\": \"10th Birthday Party\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "expected_profile": { + "age_band": "senior", + "primary_symptom": "fatigue" + }, + "critical_fields": [ + "age_band", + "primary_symptom" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "Grand Hall A", + "time": "18:00", + "reason": "Celebrating 10 years of friendship", + "name_of_event": "10th Birthday Party" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.632075548171997 + }, + { + "seed": 10003, + "task_id": "medium", + "scenario_id": "seed10003_medical_intake_medium", + "family": "medical_intake", + "request": "I need some help.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need some help.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"home\", \"time\": \"18:00\", \"reason\": \"anniversary\", \"name\": \"John\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I need some help.\", \"task_id\": \"medium\", \"task_title\": \"Moderate Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I need some help.", + "task_id": "medium", + "task_title": "Moderate Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.350191593170166 + }, + { + "seed": 10003, + "task_id": "hard", + "scenario_id": "seed10003_medical_intake_hard", + "family": "medical_intake", + "request": "I need some help.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need some help.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the description?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Park Square\", \"name\": \"Birthday Party\", \"description\": \"A celebration for my birthday\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I need some help.\", \"task_id\": \"hard\", \"task_title\": \"High Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the description?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I need some help.", + "task_id": "hard", + "task_title": "High Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.5380845069885254 + }, + { + "seed": 10004, + "task_id": "easy", + "scenario_id": "seed10004_event_planning_easy", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme or style of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of guests?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 100 people.\", \"questions_remaining\": 2, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget for the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 100 people.\", \"budget\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the name of the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the theme or style of the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of guests?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 100 people.", + "questions_remaining": 2, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the budget for the birthday party?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 100 people.\", \"budget\": \"Up to you on that one.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 1.0 + }, + "expected_profile": { + "event_type": "team-building", + "date": "today", + "guest_count": 100 + }, + "critical_fields": [ + "date", + "event_type", + "guest_count" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "today.", + "guest_count": "About 100 people.", + "budget": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "guest_count" + ] + } + } + ], + "elapsed_s": 2.1842446327209473 + }, + { + "seed": 10004, + "task_id": "medium", + "scenario_id": "seed10004_event_planning_medium", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.5097222222222222, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the guest list size?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 50 people.\", \"questions_remaining\": 4, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the venue preference?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a restaurant.\", \"questions_remaining\": 2, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme or activity preference?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"this week\", \"guest_count\": \"About 50 people.\", \"budget\": \"Up to you on that one.\", \"venue\": \"At a restaurant.\", \"theme_or_activity\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the guest list size?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 50 people.", + "questions_remaining": 4, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the budget?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the venue preference?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a restaurant.", + "questions_remaining": 2, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the theme or activity preference?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"this week\", \"guest_count\": \"About 50 people.\", \"budget\": \"Up to you on that one.\", \"venue\": \"At a restaurant.\", \"theme_or_activity\": \"Up to you on that one.\"}" + }, + "reward": 0.5097222222222222, + "done": true, + "result": { + "type": "resolution", + "score": 0.5097222222222222, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "event_type": "birthday", + "date": "this week", + "guest_count": 50, + "venue": "restaurant", + "dietary_constraints": "vegetarian" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "this week", + "guest_count": "About 50 people.", + "budget": "Up to you on that one.", + "venue": "At a restaurant.", + "theme_or_activity": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "guest_count", + "venue" + ] + } + } + ], + "elapsed_s": 2.3811328411102295 + }, + { + "seed": 10004, + "task_id": "hard", + "scenario_id": "seed10004_event_planning_hard", + "family": "event_planning", + "request": "Organize a team event.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nOrganize a team event.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a home.\", \"questions_remaining\": 4, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"themed.\", \"questions_remaining\": 3, \"field_revealed\": \"theme\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of participants?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Budget around <$100.\", \"questions_remaining\": 1, \"field_revealed\": \"budget_band\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"team_event\", \"date\": \"today.\", \"venue\": \"At a home.\", \"theme\": \"themed.\", \"participant_count\": \"Up to you on that one.\", \"budget\": \"Budget around <$100.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a home.", + "questions_remaining": 4, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the theme?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "themed.", + "questions_remaining": 3, + "field_revealed": "theme", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of participants?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the budget?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Budget around <$100.", + "questions_remaining": 1, + "field_revealed": "budget_band", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"team_event\", \"date\": \"today.\", \"venue\": \"At a home.\", \"theme\": \"themed.\", \"participant_count\": \"Up to you on that one.\", \"budget\": \"Budget around <$100.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "event_type": "team-building", + "date": "today", + "guest_count": 100, + "venue": "home", + "dietary_constraints": "vegetarian", + "budget_band": "<$100", + "theme": "themed" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "team_event", + "date": "today.", + "venue": "At a home.", + "theme": "themed.", + "participant_count": "Up to you on that one.", + "budget": "Budget around <$100." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "budget_band", + "date", + "theme", + "venue" + ] + } + } + ], + "elapsed_s": 2.1557977199554443 + }, + { + "seed": 10005, + "task_id": "easy", + "scenario_id": "seed10005_event_planning_easy", + "family": "event_planning", + "request": "Set up a celebration.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a celebration.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 4, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of guests?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 5 people.\", \"questions_remaining\": 2, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the celebration type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"this week\", \"location\": \"Up to you on that one.\", \"guest_count\": \"About 5 people.\", \"celebration_type\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 4, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of guests?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 5 people.", + "questions_remaining": 2, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the celebration type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"this week\", \"location\": \"Up to you on that one.\", \"guest_count\": \"About 5 people.\", \"celebration_type\": \"Up to you on that one.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.4444444444444444, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.75 + }, + "expected_profile": { + "event_type": "anniversary", + "guest_count": 5, + "date": "this week" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "this week", + "location": "Up to you on that one.", + "guest_count": "About 5 people.", + "celebration_type": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "guest_count" + ] + } + } + ], + "elapsed_s": 2.1309196949005127 + }, + { + "seed": 10005, + "task_id": "medium", + "scenario_id": "seed10005_event_planning_medium", + "family": "event_planning", + "request": "Set up a celebration.", + "final_score": 0.38611111111111107, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2222222222222222, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.6666666666666667 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a celebration.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the celebration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"next week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the celebration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a restaurant.\", \"questions_remaining\": 4, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme or activity of the celebration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of guests expected?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 20 people.\", \"questions_remaining\": 2, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget for the celebration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Budget around $500-2000.\", \"questions_remaining\": 1, \"field_revealed\": \"budget_band\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"celebration\", \"date\": \"next week\", \"venue\": \"At a restaurant\", \"theme\": \"Up to you on that one\", \"guest_count\": \"About 20 people\", \"budget_band\": \"Budget around $500-2000\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the celebration?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "next week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the location of the celebration?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a restaurant.", + "questions_remaining": 4, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the theme or activity of the celebration?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of guests expected?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 20 people.", + "questions_remaining": 2, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the budget for the celebration?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Budget around $500-2000.", + "questions_remaining": 1, + "field_revealed": "budget_band", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"celebration\", \"date\": \"next week\", \"venue\": \"At a restaurant\", \"theme\": \"Up to you on that one\", \"guest_count\": \"About 20 people\", \"budget_band\": \"Budget around $500-2000\"}" + }, + "reward": 0.38611111111111107, + "done": true, + "result": { + "type": "resolution", + "score": 0.38611111111111107, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2222222222222222, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.6666666666666667 + }, + "expected_profile": { + "event_type": "anniversary", + "date": "next week", + "guest_count": 20, + "venue": "restaurant", + "budget_band": "$500-2000" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "celebration", + "date": "next week", + "venue": "At a restaurant", + "theme": "Up to you on that one", + "guest_count": "About 20 people", + "budget_band": "Budget around $500-2000" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "budget_band", + "date", + "guest_count", + "venue" + ] + } + } + ], + "elapsed_s": 2.6591756343841553 + }, + { + "seed": 10005, + "task_id": "hard", + "scenario_id": "seed10005_event_planning_hard", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.29500000000000004, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the guest list size?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 20 people.\", \"questions_remaining\": 4, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Budget around <$100.\", \"questions_remaining\": 3, \"field_revealed\": \"budget_band\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location preference?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a rented hall.\", \"questions_remaining\": 2, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 20 people.\", \"budget_band\": \"Budget around <$100.\", \"venue\": \"At a rented hall.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the guest list size?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 20 people.", + "questions_remaining": 4, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the budget?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Budget around <$100.", + "questions_remaining": 3, + "field_revealed": "budget_band", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location preference?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a rented hall.", + "questions_remaining": 2, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"today.\", \"guest_count\": \"About 20 people.\", \"budget_band\": \"Budget around <$100.\", \"venue\": \"At a rented hall.\"}" + }, + "reward": 0.29500000000000004, + "done": true, + "result": { + "type": "resolution", + "score": 0.29500000000000004, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "event_type": "anniversary", + "date": "today", + "guest_count": 20, + "venue": "rented hall", + "budget_band": "<$100", + "theme": "themed", + "dietary_constraints": "none" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "today.", + "guest_count": "About 20 people.", + "budget_band": "Budget around <$100.", + "venue": "At a rented hall." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "budget_band", + "date", + "guest_count", + "venue" + ] + } + } + ], + "elapsed_s": 2.1525380611419678 + }, + { + "seed": 10006, + "task_id": "easy", + "scenario_id": "seed10006_medical_intake_easy", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the health issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the specific health issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"health_issue\", \"date\": \"2024-12-25\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\", \"specific_issue\": \"unknown\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the health issue?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the specific health issue?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"health_issue\", \"date\": \"2024-12-25\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\", \"specific_issue\": \"unknown\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "duration": "chronic", + "severity": "severe" + }, + "critical_fields": [ + "duration", + "severity" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "health_issue", + "date": "2024-12-25", + "location": "unknown", + "time": "unknown", + "reason": "unknown", + "specific_issue": "unknown" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.4495530128479004 + }, + { + "seed": 10006, + "task_id": "medium", + "scenario_id": "seed10006_medical_intake_medium", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What specific health issue are you experiencing?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What specific symptoms are you experiencing?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"It's a rash.\", \"questions_remaining\": 4, \"field_revealed\": \"primary_symptom\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the rash?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the duration of the rash?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About chronic.\", \"questions_remaining\": 2, \"field_revealed\": \"duration\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the cause of the rash?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the possible cause of the rash?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"health_issue\", \"primary_symptom\": \"rash\", \"location\": \"unknown\", \"duration\": \"chronic\", \"cause\": \"unknown\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What specific health issue are you experiencing?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What specific symptoms are you experiencing?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "It's a rash.", + "questions_remaining": 4, + "field_revealed": "primary_symptom", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the rash?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the duration of the rash?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About chronic.", + "questions_remaining": 2, + "field_revealed": "duration", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the cause of the rash?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the possible cause of the rash?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"health_issue\", \"primary_symptom\": \"rash\", \"location\": \"unknown\", \"duration\": \"chronic\", \"cause\": \"unknown\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "primary_symptom": "rash", + "duration": "chronic", + "severity": "severe", + "age_band": "teen" + }, + "critical_fields": [ + "age_band", + "duration", + "primary_symptom", + "severity" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "health_issue", + "primary_symptom": "rash", + "location": "unknown", + "duration": "chronic", + "cause": "unknown" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [ + "duration", + "primary_symptom" + ] + } + } + ], + "elapsed_s": 2.417412757873535 + }, + { + "seed": 10006, + "task_id": "hard", + "scenario_id": "seed10006_medical_intake_hard", + "family": "medical_intake", + "request": "I need some help.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need some help.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Park\", \"name\": \"Happy Birthday\", \"reason\": \"To celebrate\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I need some help.\", \"task_id\": \"hard\", \"task_title\": \"High Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I need some help.", + "task_id": "hard", + "task_title": "High Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.4852147102355957 + }, + { + "seed": 10007, + "task_id": "easy", + "scenario_id": "seed10007_support_triage_easy", + "family": "support_triage", + "request": "I need help with an order.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need help with an order.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the contact information for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the delivery or pickup time for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "get_task_info()" + }, + { + "role": "user", + "content": "{\"request\": \"I need help with an order.\", \"task_id\": \"easy\", \"task_title\": \"Mild Ambiguity\", \"family\": \"support_triage\", \"questions_remaining\": 1, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"order\", \"date\": \"unknown\", \"location\": \"unknown\", \"contact\": \"unknown\", \"delivery_time\": \"unknown\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the contact information for the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the delivery or pickup time for the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I need help with an order.", + "task_id": "easy", + "task_title": "Mild Ambiguity", + "family": "support_triage", + "questions_remaining": 1, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"order\", \"date\": \"unknown\", \"location\": \"unknown\", \"contact\": \"unknown\", \"delivery_time\": \"unknown\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "item_issue": "missing", + "refund_or_replace": "unsure", + "order_id": "#1199" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "order", + "date": "unknown", + "location": "unknown", + "contact": "unknown", + "delivery_time": "unknown" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.2636802196502686 + }, + { + "seed": 10007, + "task_id": "medium", + "scenario_id": "seed10007_support_triage_medium", + "family": "support_triage", + "request": "There's an issue with my purchase.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nThere's an issue with my purchase.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the issue with the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the expected resolution for the issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'd prefer a store-credit.\", \"questions_remaining\": 1, \"field_revealed\": \"refund_or_replace\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"Store A\", \"reason\": \"issue with purchase\", \"resolution\": \"store-credit\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the reason for the issue with the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the expected resolution for the issue?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "I'd prefer a store-credit.", + "questions_remaining": 1, + "field_revealed": "refund_or_replace", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"Store A\", \"reason\": \"issue with purchase\", \"resolution\": \"store-credit\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "order_id": "#1199", + "item_issue": "late", + "refund_or_replace": "store-credit", + "urgency": "medium", + "channel_preferred": "email" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "purchase", + "date": "2024-12-25", + "location": "Store A", + "reason": "issue with purchase", + "resolution": "store-credit" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "refund_or_replace" + ] + } + } + ], + "elapsed_s": 2.3607757091522217 + }, + { + "seed": 10007, + "task_id": "hard", + "scenario_id": "seed10007_support_triage_hard", + "family": "support_triage", + "request": "There's an issue with my purchase.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nThere's an issue with my purchase.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the issue with the purchase?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the expected resolution for the issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'd prefer a store-credit.\", \"questions_remaining\": 1, \"field_revealed\": \"refund_or_replace\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"Store A\", \"reason\": \"issue with purchase\", \"resolution\": \"store-credit\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the reason for the issue with the purchase?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the expected resolution for the issue?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "I'd prefer a store-credit.", + "questions_remaining": 1, + "field_revealed": "refund_or_replace", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"purchase\", \"date\": \"2024-12-25\", \"location\": \"Store A\", \"reason\": \"issue with purchase\", \"resolution\": \"store-credit\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "order_id": "#1199", + "item_issue": "late", + "refund_or_replace": "store-credit", + "urgency": "medium", + "channel_preferred": "email" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "purchase", + "date": "2024-12-25", + "location": "Store A", + "reason": "issue with purchase", + "resolution": "store-credit" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "refund_or_replace" + ] + } + } + ], + "elapsed_s": 2.425291061401367 + }, + { + "seed": 10008, + "task_id": "easy", + "scenario_id": "seed10008_medical_intake_easy", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the health issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the specific health issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"health_issue\", \"date\": \"unknown\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\", \"specific_issue\": \"unknown\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the health issue?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the specific health issue?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"health_issue\", \"date\": \"unknown\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\", \"specific_issue\": \"unknown\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "age_band": "adult", + "duration": "chronic", + "severity": "mild" + }, + "critical_fields": [ + "age_band", + "duration", + "severity" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "health_issue", + "date": "unknown", + "location": "unknown", + "time": "unknown", + "reason": "unknown", + "specific_issue": "unknown" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.272193670272827 + }, + { + "seed": 10008, + "task_id": "medium", + "scenario_id": "seed10008_medical_intake_medium", + "family": "medical_intake", + "request": "I have a problem.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI have a problem.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Grand Hall A\", \"time\": \"18:00\", \"reason\": \"Celebrating 10 years of service\", \"name\": \"10th Anniversary\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I have a problem.\", \"task_id\": \"medium\", \"task_title\": \"Moderate Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I have a problem.", + "task_id": "medium", + "task_title": "Moderate Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.6440603733062744 + }, + { + "seed": 10008, + "task_id": "hard", + "scenario_id": "seed10008_medical_intake_hard", + "family": "medical_intake", + "request": "I need some help.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need some help.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"home\", \"time\": \"6:00 PM\", \"reason\": \"anniversary\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"home\", \"time\": \"6:00 PM\", \"reason\": \"anniversary\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "primary_symptom": "cough", + "duration": "chronic", + "severity": "mild", + "prior_conditions": "other", + "medications": "prescription", + "age_band": "adult" + }, + "critical_fields": [ + "age_band", + "duration", + "primary_symptom", + "severity" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "home", + "time": "6:00 PM", + "reason": "anniversary" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.282386302947998 + }, + { + "seed": 10009, + "task_id": "easy", + "scenario_id": "seed10009_support_triage_easy", + "family": "support_triage", + "request": "I need help with an order.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need help with an order.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Does the order require any specific services or products?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is there any additional information required to complete the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the expected delivery or pickup time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Are there any special instructions or preferences for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Order #9027.\", \"questions_remaining\": 1, \"field_revealed\": \"order_id\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"order_id\": \"Order #9027.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "Does the order require any specific services or products?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "Is there any additional information required to complete the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the expected delivery or pickup time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "Are there any special instructions or preferences for the order?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Order #9027.", + "questions_remaining": 1, + "field_revealed": "order_id", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"order_id\": \"Order #9027.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "refund_or_replace": "refund", + "order_id": "#9027" + }, + "critical_fields": [ + "order_id", + "refund_or_replace" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "order_id": "Order #9027." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "order_id" + ] + } + } + ], + "elapsed_s": 1.8414006233215332 + }, + { + "seed": 10009, + "task_id": "medium", + "scenario_id": "seed10009_support_triage_medium", + "family": "support_triage", + "request": "Something went wrong.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething went wrong.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Grand Hall\", \"reason\": \"Anniversary\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"time\": \"18:00\", \"location\": \"Grand Hall\", \"reason\": \"Anniversary\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "order_id": "#4521", + "item_issue": "wrong-item", + "refund_or_replace": "unsure", + "channel_preferred": "phone" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "time": "18:00", + "location": "Grand Hall", + "reason": "Anniversary" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.060541868209839 + }, + { + "seed": 10009, + "task_id": "hard", + "scenario_id": "seed10009_support_triage_hard", + "family": "support_triage", + "request": "My order is wrong.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nMy order is wrong.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the order being wrong?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"late.\", \"questions_remaining\": 1, \"field_revealed\": \"item_issue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Grand Hall A\", \"time\": \"18:00\", \"reason_for_order_wrong\": \"late\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the order being wrong?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "late.", + "questions_remaining": 1, + "field_revealed": "item_issue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Grand Hall A\", \"time\": \"18:00\", \"reason_for_order_wrong\": \"late\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "order_id": "#4521", + "item_issue": "late", + "refund_or_replace": "unsure", + "channel_preferred": "phone", + "urgency": "medium" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "Grand Hall A", + "time": "18:00", + "reason_for_order_wrong": "late" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "item_issue" + ] + } + } + ], + "elapsed_s": 2.2882487773895264 + }, + { + "seed": 10010, + "task_id": "easy", + "scenario_id": "seed10010_event_planning_easy", + "family": "event_planning", + "request": "Set up a celebration.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a celebration.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"next week.\", \"questions_remaining\": 4, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of guests?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 10 people.\", \"questions_remaining\": 1, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"celebration\", \"date\": \"next week\", \"location\": \"Up to you on that one.\", \"theme\": \"Up to you on that one.\", \"guest_count\": \"About 10 people.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "next week.", + "questions_remaining": 4, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the theme?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the number of guests?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 10 people.", + "questions_remaining": 1, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"celebration\", \"date\": \"next week\", \"location\": \"Up to you on that one.\", \"theme\": \"Up to you on that one.\", \"guest_count\": \"About 10 people.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "event_type": "anniversary", + "guest_count": 10, + "date": "next week" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "celebration", + "date": "next week", + "location": "Up to you on that one.", + "theme": "Up to you on that one.", + "guest_count": "About 10 people." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "guest_count" + ] + } + } + ], + "elapsed_s": 2.16644024848938 + }, + { + "seed": 10010, + "task_id": "medium", + "scenario_id": "seed10010_event_planning_medium", + "family": "event_planning", + "request": "Set up a celebration.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a celebration.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 4, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a home.\", \"questions_remaining\": 3, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the duration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Up to you on that one.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"celebration\", \"date\": \"this week\", \"venue\": \"At a home\", \"theme\": \"Up to you on that one.\", \"duration\": \"Up to you on that one.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 4, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a home.", + "questions_remaining": 3, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the theme?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the duration?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "Up to you on that one.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"celebration\", \"date\": \"this week\", \"venue\": \"At a home\", \"theme\": \"Up to you on that one.\", \"duration\": \"Up to you on that one.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8 + }, + "expected_profile": { + "event_type": "anniversary", + "date": "this week", + "guest_count": 20, + "venue": "home", + "budget_band": "$2000+" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "celebration", + "date": "this week", + "venue": "At a home", + "theme": "Up to you on that one.", + "duration": "Up to you on that one." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "venue" + ] + } + } + ], + "elapsed_s": 2.0731544494628906 + }, + { + "seed": 10010, + "task_id": "hard", + "scenario_id": "seed10010_event_planning_hard", + "family": "event_planning", + "request": "Plan a birthday party.", + "final_score": 0.4363636363636364, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2727272727272727, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8333333333333334 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nPlan a birthday party.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"tomorrow.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"At a home.\", \"questions_remaining\": 4, \"field_revealed\": \"venue\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the guest list size?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"About 50 people.\", \"questions_remaining\": 3, \"field_revealed\": \"guest_count\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the theme or decoration style of the party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"casual.\", \"questions_remaining\": 2, \"field_revealed\": \"theme\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the budget for the birthday party?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Budget around <$100.\", \"questions_remaining\": 1, \"field_revealed\": \"budget_band\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"tomorrow\", \"venue\": \"At a home\", \"guest_count\": \"About 50 people\", \"theme\": \"casual\", \"budget_band\": \"Budget around <$100.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "tomorrow.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the location of the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "At a home.", + "questions_remaining": 4, + "field_revealed": "venue", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the guest list size?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "About 50 people.", + "questions_remaining": 3, + "field_revealed": "guest_count", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the theme or decoration style of the party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "casual.", + "questions_remaining": 2, + "field_revealed": "theme", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the budget for the birthday party?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Budget around <$100.", + "questions_remaining": 1, + "field_revealed": "budget_band", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"tomorrow\", \"venue\": \"At a home\", \"guest_count\": \"About 50 people\", \"theme\": \"casual\", \"budget_band\": \"Budget around <$100.\"}" + }, + "reward": 0.4363636363636364, + "done": true, + "result": { + "type": "resolution", + "score": 0.4363636363636364, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.2727272727272727, + "InfoGainRubric": 0.75, + "QuestionEfficiencyRubric": 0.16666666666666663, + "HallucinationCheckRubric": 0.8333333333333334 + }, + "expected_profile": { + "event_type": "anniversary", + "date": "tomorrow", + "guest_count": 50, + "venue": "home", + "budget_band": "<$100", + "theme": "casual", + "dietary_constraints": "mixed" + }, + "critical_fields": [ + "date", + "event_type", + "guest_count", + "venue" + ], + "required_keys": [ + "event_type", + "date", + "guest_count", + "venue" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "tomorrow", + "venue": "At a home", + "guest_count": "About 50 people", + "theme": "casual", + "budget_band": "Budget around <$100." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "budget_band", + "date", + "guest_count", + "theme", + "venue" + ] + } + } + ], + "elapsed_s": 2.526638984680176 + }, + { + "seed": 10011, + "task_id": "easy", + "scenario_id": "seed10011_medical_intake_easy", + "family": "medical_intake", + "request": "I'm not feeling well.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI'm not feeling well.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Unknown\", \"time\": \"Unknown\", \"name\": \"Unknown\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I'm not feeling well.\", \"task_id\": \"easy\", \"task_title\": \"Mild Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I'm not feeling well.", + "task_id": "easy", + "task_title": "Mild Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.5107436180114746 + }, + { + "seed": 10011, + "task_id": "medium", + "scenario_id": "seed10011_medical_intake_medium", + "family": "medical_intake", + "request": "I'm not feeling well.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI'm not feeling well.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Unknown\", \"time\": \"Unknown\", \"reason\": \"Unknown\", \"name\": \"Unknown\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I'm not feeling well.\", \"task_id\": \"medium\", \"task_title\": \"Moderate Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I'm not feeling well.", + "task_id": "medium", + "task_title": "Moderate Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.3780229091644287 + }, + { + "seed": 10011, + "task_id": "hard", + "scenario_id": "seed10011_medical_intake_hard", + "family": "medical_intake", + "request": "I have a problem.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI have a problem.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Unknown\", \"time\": \"Unknown\", \"reason\": \"Unknown\", \"name\": \"Unknown\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"I have a problem.\", \"task_id\": \"hard\", \"task_title\": \"High Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I have a problem.", + "task_id": "hard", + "task_title": "High Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.3850440979003906 + }, + { + "seed": 10012, + "task_id": "easy", + "scenario_id": "seed10012_meeting_scheduling_easy", + "family": "meeting_scheduling", + "request": "Schedule a sync.", + "final_score": 0.3333333333333333, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 4, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSchedule a sync.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of participants?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"whole team.\", \"questions_remaining\": 2, \"field_revealed\": \"participants\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"sync\", \"date\": \"2024-12-25\", \"time\": \"10:00 AM\", \"location\": \"Conference Room A\", \"participants\": \"whole team\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of participants?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "whole team.", + "questions_remaining": 2, + "field_revealed": "participants", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"sync\", \"date\": \"2024-12-25\", \"time\": \"10:00 AM\", \"location\": \"Conference Room A\", \"participants\": \"whole team\"}" + }, + "reward": 0.3333333333333333, + "done": true, + "result": { + "type": "resolution", + "score": 0.3333333333333333, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "whole team", + "duration_minutes": 45, + "time": "afternoon" + }, + "critical_fields": [ + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "sync", + "date": "2024-12-25", + "time": "10:00 AM", + "location": "Conference Room A", + "participants": "whole team" + }, + "parse_error": null, + "questions_asked": 4, + "fields_revealed": [ + "participants" + ] + } + } + ], + "elapsed_s": 1.9142417907714844 + }, + { + "seed": 10012, + "task_id": "medium", + "scenario_id": "seed10012_meeting_scheduling_medium", + "family": "meeting_scheduling", + "request": "Set up a meeting.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a meeting.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the duration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"30 minutes.\", \"questions_remaining\": 2, \"field_revealed\": \"duration_minutes\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the person?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"meeting\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"duration_minutes\": \"30\", \"person\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the duration?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "30 minutes.", + "questions_remaining": 2, + "field_revealed": "duration_minutes", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the person?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"meeting\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"duration_minutes\": \"30\", \"person\": \"No preference, you choose.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "whole team", + "date": "today", + "time": "evening", + "duration_minutes": 30, + "platform": "google-meet" + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "meeting", + "date": "today.", + "time": "No preference, you choose.", + "location": "No preference, you choose.", + "duration_minutes": "30", + "person": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date", + "duration_minutes" + ] + } + } + ], + "elapsed_s": 2.2221150398254395 + }, + { + "seed": 10012, + "task_id": "hard", + "scenario_id": "seed10012_meeting_scheduling_hard", + "family": "meeting_scheduling", + "request": "Set up a meeting.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a meeting.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the participant count for the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the agenda for the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"meeting\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"participant_count\": \"No preference, you choose.\", \"agenda\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the meeting?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the participant count for the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the agenda for the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"meeting\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"participant_count\": \"No preference, you choose.\", \"agenda\": \"No preference, you choose.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "whole team", + "date": "today", + "time": "evening", + "duration_minutes": 30, + "platform": "google-meet" + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "meeting", + "date": "today.", + "time": "No preference, you choose.", + "location": "No preference, you choose.", + "participant_count": "No preference, you choose.", + "agenda": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date" + ] + } + } + ], + "elapsed_s": 2.557643413543701 + }, + { + "seed": 10013, + "task_id": "easy", + "scenario_id": "seed10013_support_triage_easy", + "family": "support_triage", + "request": "I need help with an order.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need help with an order.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the contact information for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the purpose of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the expected delivery date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"New York, NY\", \"contact\": \"john.doe@example.com\", \"purpose\": \"Celebration\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the contact information for the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the purpose of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the expected delivery date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"New York, NY\", \"contact\": \"john.doe@example.com\", \"purpose\": \"Celebration\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "order_id": "#7830", + "refund_or_replace": "store-credit", + "item_issue": "late" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "New York, NY", + "contact": "john.doe@example.com", + "purpose": "Celebration" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.6081066131591797 + }, + { + "seed": 10013, + "task_id": "medium", + "scenario_id": "seed10013_support_triage_medium", + "family": "support_triage", + "request": "I need help with an order.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need help with an order.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Does the order require any specific services or items?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Is there a specific time frame or deadline for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Are there any specific requirements or constraints for the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"Does the order require any additional documentation or proof?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "get_task_info()" + }, + { + "role": "user", + "content": "{\"request\": \"I need help with an order.\", \"task_id\": \"medium\", \"task_title\": \"Moderate Ambiguity\", \"family\": \"support_triage\", \"questions_remaining\": 1, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"order\", \"date\": \"2024-12-25\", \"services\": \"delivery\", \"deadline\": \"2024-12-30\", \"documentation\": \"invoice\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "Does the order require any specific services or items?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "Is there a specific time frame or deadline for the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "Are there any specific requirements or constraints for the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "Does the order require any additional documentation or proof?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "I need help with an order.", + "task_id": "medium", + "task_title": "Moderate Ambiguity", + "family": "support_triage", + "questions_remaining": 1, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"order\", \"date\": \"2024-12-25\", \"services\": \"delivery\", \"deadline\": \"2024-12-30\", \"documentation\": \"invoice\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "order_id": "#7830", + "item_issue": "never-arrived", + "refund_or_replace": "store-credit", + "urgency": "medium", + "channel_preferred": "chat" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "order", + "date": "2024-12-25", + "services": "delivery", + "deadline": "2024-12-30", + "documentation": "invoice" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.701340913772583 + }, + { + "seed": 10013, + "task_id": "hard", + "scenario_id": "seed10013_support_triage_hard", + "family": "support_triage", + "request": "I need help with an order.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nI need help with an order.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the delivery method of the order?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the customer's name?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I don't really know \\u2014 does it matter?\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the order number?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"Order #7830.\", \"questions_remaining\": 0, \"field_revealed\": \"order_id\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"New York\", \"delivery_method\": \"pickup\", \"order_number\": \"Order #7830\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the delivery method of the order?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the customer's name?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I don't really know \u2014 does it matter?", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the order number?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "Order #7830.", + "questions_remaining": 0, + "field_revealed": "order_id", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"New York\", \"delivery_method\": \"pickup\", \"order_number\": \"Order #7830\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "order_id": "#7830", + "item_issue": "never-arrived", + "refund_or_replace": "store-credit", + "urgency": "medium", + "channel_preferred": "chat" + }, + "critical_fields": [ + "item_issue", + "order_id", + "refund_or_replace", + "urgency" + ], + "required_keys": [ + "order_id", + "item_issue", + "refund_or_replace" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "New York", + "delivery_method": "pickup", + "order_number": "Order #7830" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [ + "order_id" + ] + } + } + ], + "elapsed_s": 2.6317193508148193 + }, + { + "seed": 10014, + "task_id": "easy", + "scenario_id": "seed10014_meeting_scheduling_easy", + "family": "meeting_scheduling", + "request": "Set up a meeting.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSet up a meeting.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the participant count?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the agenda for the meeting?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"meeting\", \"date\": \"2024-12-25\", \"time\": \"14:00\", \"location\": \"Conference Room 1\", \"participant_count\": \"5\", \"agenda\": \"Discussion on project updates\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the participant count?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the agenda for the meeting?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"meeting\", \"date\": \"2024-12-25\", \"time\": \"14:00\", \"location\": \"Conference Room 1\", \"participant_count\": \"5\", \"agenda\": \"Discussion on project updates\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "time": "flexible", + "participants": "whole team" + }, + "critical_fields": [ + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "meeting", + "date": "2024-12-25", + "time": "14:00", + "location": "Conference Room 1", + "participant_count": "5", + "agenda": "Discussion on project updates" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.454289197921753 + }, + { + "seed": 10014, + "task_id": "medium", + "scenario_id": "seed10014_meeting_scheduling_medium", + "family": "meeting_scheduling", + "request": "Get a call on the calendar.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nGet a call on the calendar.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"next week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the meeting duration?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"next week\", \"time\": \"No preference, you choose.\", \"duration\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"event\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "next week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the meeting duration?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"next week\", \"time\": \"No preference, you choose.\", \"duration\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"event\": \"No preference, you choose.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "just me and X", + "date": "next week", + "time": "morning", + "platform": "phone" + }, + "critical_fields": [ + "date", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "next week", + "time": "No preference, you choose.", + "duration": "No preference, you choose.", + "location": "No preference, you choose.", + "event": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date" + ] + } + } + ], + "elapsed_s": 2.2367234230041504 + }, + { + "seed": 10014, + "task_id": "hard", + "scenario_id": "seed10014_meeting_scheduling_hard", + "family": "meeting_scheduling", + "request": "Get a call on the calendar.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nGet a call on the calendar.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"today.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event name?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "today.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the event name?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"today.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "external client", + "date": "today", + "time": "flexible", + "platform": "google-meet", + "duration_minutes": 30 + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "today.", + "time": "No preference, you choose.", + "location": "No preference, you choose." + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [ + "date" + ] + } + } + ], + "elapsed_s": 1.8531725406646729 + }, + { + "seed": 10015, + "task_id": "easy", + "scenario_id": "seed10015_medical_intake_easy", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 6, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the health issue?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the health issue?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "age_band": "teen", + "duration": "chronic" + }, + "critical_fields": [ + "age_band", + "duration" + ], + "required_keys": [ + "primary_symptom", + "duration", + "severity" + ], + "submitted_plan": { + "event_type": "birthday", + "date": "2024-12-25", + "location": "unknown", + "time": "unknown", + "reason": "unknown" + }, + "parse_error": null, + "questions_asked": 6, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.447335958480835 + }, + { + "seed": 10015, + "task_id": "medium", + "scenario_id": "seed10015_medical_intake_medium", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"Unknown\", \"time\": \"Unknown\", \"name\": \"Unknown\", \"reason\": \"Unknown\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"Something's off with my health.\", \"task_id\": \"medium\", \"task_title\": \"Moderate Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "Something's off with my health.", + "task_id": "medium", + "task_title": "Moderate Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.6541805267333984 + }, + { + "seed": 10015, + "task_id": "hard", + "scenario_id": "seed10015_medical_intake_hard", + "family": "medical_intake", + "request": "Something's off with my health.", + "final_score": 0.0, + "score_breakdown": {}, + "questions_asked": 7, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSomething's off with my health.\n\nYou have 12 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event type?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the reason for the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the name of the event?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"I'm not sure about that, sorry.\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\", \"location\": \"unknown\", \"time\": \"unknown\", \"reason\": \"unknown\", \"name\": \"unknown\"}')" + }, + { + "role": "user", + "content": "{\"request\": \"Something's off with my health.\", \"task_id\": \"hard\", \"task_title\": \"High Ambiguity\", \"family\": \"medical_intake\", \"questions_remaining\": 0, \"instructions\": \"Ask clarifying questions via ask_question(question) \\u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about.\"}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "user", + "content": "{\"answer\": \"(no more questions allowed)\", \"questions_remaining\": 0, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": true}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the event type?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the time of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the reason for the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "ask_question", + "args": { + "question": "What is the name of the event?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "I'm not sure about that, sorry.", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 6, + "tool": "get_task_info", + "args": {}, + "reward": 0.0, + "done": false, + "result": { + "request": "Something's off with my health.", + "task_id": "hard", + "task_title": "High Ambiguity", + "family": "medical_intake", + "questions_remaining": 0, + "instructions": "Ask clarifying questions via ask_question(question) \u2014 you have a 6-question budget. Then submit your final plan via propose_plan(plan) where plan is a JSON string object containing the required keys for the task family. Avoid hallucinating values for fields you never asked about." + } + }, + { + "step": 7, + "tool": "ask_question", + "args": { + "question": "What is the main requirement?" + }, + "reward": -0.05, + "done": true, + "result": { + "answer": "(no more questions allowed)", + "questions_remaining": 0, + "field_revealed": null, + "duplicate": false, + "over_cap": true + } + } + ], + "elapsed_s": 2.550715684890747 + }, + { + "seed": 10016, + "task_id": "easy", + "scenario_id": "seed10016_meeting_scheduling_easy", + "family": "meeting_scheduling", + "request": "Get a call on the calendar.", + "final_score": 0.0, + "score_breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "questions_asked": 5, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nGet a call on the calendar.\n\nYou have 8 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 5, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the event name?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 2, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the contact information?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 1, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"call\", \"date\": \"2024-12-25\", \"time\": \"14:00\", \"location\": \"Conference Room 1\", \"event_name\": \"Team Meeting\", \"contact_info\": \"john.doe@example.com\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 5, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the event name?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 2, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "ask_question", + "args": { + "question": "What is the contact information?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 1, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 5, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"call\", \"date\": \"2024-12-25\", \"time\": \"14:00\", \"location\": \"Conference Room 1\", \"event_name\": \"Team Meeting\", \"contact_info\": \"john.doe@example.com\"}" + }, + "reward": 0.0, + "done": true, + "result": { + "type": "resolution", + "score": 0.0, + "breakdown": { + "FormatCheckRubric": 0.0, + "FieldMatchRubric": 0.3333333333333333, + "InfoGainRubric": 0.3333333333333333, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.33333333333333337 + }, + "expected_profile": { + "participants": "just me and X", + "time": "evening", + "duration_minutes": 90 + }, + "critical_fields": [ + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "call", + "date": "2024-12-25", + "time": "14:00", + "location": "Conference Room 1", + "event_name": "Team Meeting", + "contact_info": "john.doe@example.com" + }, + "parse_error": null, + "questions_asked": 5, + "fields_revealed": [] + } + } + ], + "elapsed_s": 2.331054925918579 + }, + { + "seed": 10016, + "task_id": "medium", + "scenario_id": "seed10016_meeting_scheduling_medium", + "family": "meeting_scheduling", + "request": "Schedule a sync.", + "final_score": 0.25, + "score_breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.5, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.6666666666666667 + }, + "questions_asked": 4, + "format_pass": null, + "parse_error": null, + "messages": [ + { + "role": "system", + "content": "You are a helpful assistant that books and plans things for users.\nThe user's request will be intentionally ambiguous \u2014 you do NOT yet have all the information needed to make a good plan.\n\nYou have three tools:\n - ask_question(question): ask the user ONE targeted clarifying question (max 6 across the episode).\n - propose_plan(plan): submit your final plan as a JSON STRING with the required fields. This ENDS the episode.\n - get_task_info(): re-read the original user request.\n\nStrategy:\n 1. Read the required plan fields listed in the task description.\n 2. Use ask_question to ask about EACH required field you do not already know.\n 3. When you have enough info, call propose_plan with a JSON string containing ALL required fields.\n\nRules:\n - Be efficient. Each unnecessary question costs reward.\n - Your plan MUST include every required field listed in the task. Missing fields score zero.\n - NEVER include fields in your plan that you weren't told about. No hallucinating values.\n - The `plan` argument MUST be a JSON STRING (not a dict). Use the exact field names from the required fields list.\n" + }, + { + "role": "user", + "content": "USER REQUEST:\nSchedule a sync.\n\nYou have 10 steps. Available tools: ask_question(question), propose_plan(plan), get_task_info().\n\nRESPONSE FORMAT: Reply with ONE function call only, no other text.\nExamples:\n ask_question(\"What is the date?\")\n propose_plan('{\"event_type\": \"birthday\", \"date\": \"2024-12-25\"}')\n get_task_info()\n" + }, + { + "role": "assistant", + "content": "ask_question(\"What is the date of the sync?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"this week.\", \"questions_remaining\": 5, \"field_revealed\": \"date\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the time of the sync?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 4, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the location of the sync?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"No preference, you choose.\", \"questions_remaining\": 3, \"field_revealed\": null, \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "ask_question(\"What is the number of participants?\")" + }, + { + "role": "user", + "content": "{\"answer\": \"just me and X.\", \"questions_remaining\": 2, \"field_revealed\": \"participants\", \"duplicate\": false, \"over_cap\": false}\n\nReminder: Reply with ONE function call only (ask_question/propose_plan/get_task_info), no other text." + }, + { + "role": "assistant", + "content": "propose_plan('{\"event_type\": \"sync\", \"date\": \"this week.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"participants\": \"just me and X.\"}')" + } + ], + "trace": [ + { + "step": 0, + "tool": "ask_question", + "args": { + "question": "What is the date of the sync?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "this week.", + "questions_remaining": 5, + "field_revealed": "date", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 1, + "tool": "ask_question", + "args": { + "question": "What is the time of the sync?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 4, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 2, + "tool": "ask_question", + "args": { + "question": "What is the location of the sync?" + }, + "reward": 0.02, + "done": false, + "result": { + "answer": "No preference, you choose.", + "questions_remaining": 3, + "field_revealed": null, + "duplicate": false, + "over_cap": false + } + }, + { + "step": 3, + "tool": "ask_question", + "args": { + "question": "What is the number of participants?" + }, + "reward": 0.05, + "done": false, + "result": { + "answer": "just me and X.", + "questions_remaining": 2, + "field_revealed": "participants", + "duplicate": false, + "over_cap": false + } + }, + { + "step": 4, + "tool": "propose_plan", + "args": { + "plan": "{\"event_type\": \"sync\", \"date\": \"this week.\", \"time\": \"No preference, you choose.\", \"location\": \"No preference, you choose.\", \"participants\": \"just me and X.\"}" + }, + "reward": 0.25, + "done": true, + "result": { + "type": "resolution", + "score": 0.25, + "breakdown": { + "FormatCheckRubric": 1.0, + "FieldMatchRubric": 0.0, + "InfoGainRubric": 0.5, + "QuestionEfficiencyRubric": 0.33333333333333337, + "HallucinationCheckRubric": 0.6666666666666667 + }, + "expected_profile": { + "participants": "just me and X", + "date": "this week", + "time": "evening", + "duration_minutes": 30, + "platform": "google-meet" + }, + "critical_fields": [ + "date", + "duration_minutes", + "participants", + "time" + ], + "required_keys": [ + "participants", + "date", + "time" + ], + "submitted_plan": { + "event_type": "sync", + "date": "this week.", + "time": "No preference, you choose.", + "location": "No preference, you choose.", + "participants": "just me and X." + }, + "parse_error": null, + "questions_asked": 4, + "fields_revealed": [ + "date", + "participants" + ] + } + } + ], + "elapsed_s": 1.956272840499878 + } + ] +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..9787484 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,12 @@ +{ + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.6, + "top_k": 20, + "top_p": 0.95, + "transformers_version": "5.7.0.dev0" +} diff --git a/log_history.json b/log_history.json new file mode 100644 index 0000000..2fc23e0 --- /dev/null +++ b/log_history.json @@ -0,0 +1,14011 @@ +[ + { + "loss": -0.14565494656562805, + "grad_norm": 2.059197187423706, + "learning_rate": 0.0, + "num_tokens": 10568.0, + "completions/mean_length": 643.25, + "completions/min_length": 180.0, + "completions/max_length": 924.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 547.5, + "completions/min_terminated_length": 180.0, + "completions/max_terminated_length": 924.0, + "tools/call_frequency": 24.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": -0.04374999925494194, + "rewards/reward_func/std": 0.07763238251209259, + "reward": -0.04374999925494194, + "reward_std": 0.077632375061512, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0012343216221779585, + "sampling/sampling_logp_difference/max": 0.674403190612793, + "sampling/importance_sampling_ratio/min": 0.4104541838169098, + "sampling/importance_sampling_ratio/mean": 1.0825542211532593, + "sampling/importance_sampling_ratio/max": 2.163515090942383, + "kl": 0.0, + "entropy": 0.017752465690136887, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 201.99276200681925, + "epoch": 1.953125e-05, + "step": 1 + }, + { + "loss": -0.13430844247341156, + "grad_norm": 1.4767537117004395, + "learning_rate": 1e-07, + "num_tokens": 21839.0, + "completions/mean_length": 723.75, + "completions/min_length": 143.0, + "completions/max_length": 925.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 619.4000244140625, + "completions/min_terminated_length": 143.0, + "completions/max_terminated_length": 925.0, + "tools/call_frequency": 17.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.05937500298023224, + "rewards/reward_func/std": 0.16253434121608734, + "reward": 0.05937500298023224, + "reward_std": 0.16253434121608734, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001157777151092887, + "sampling/sampling_logp_difference/max": 0.37639790773391724, + "sampling/importance_sampling_ratio/min": 0.5049796104431152, + "sampling/importance_sampling_ratio/mean": 1.198677659034729, + "sampling/importance_sampling_ratio/max": 2.195049524307251, + "kl": 0.0, + "entropy": 0.01981475652428344, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.23118992894888, + "epoch": 3.90625e-05, + "step": 2 + }, + { + "loss": 0.0334477536380291, + "grad_norm": 2.290715217590332, + "learning_rate": 2e-07, + "num_tokens": 32435.0, + "completions/mean_length": 639.875, + "completions/min_length": 140.0, + "completions/max_length": 923.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 486.20001220703125, + "completions/min_terminated_length": 140.0, + "completions/max_terminated_length": 923.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.08958333730697632, + "rewards/reward_func/std": 0.24282360076904297, + "reward": 0.08958333730697632, + "reward_std": 0.24282360076904297, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011750324629247189, + "sampling/sampling_logp_difference/max": 0.3143371343612671, + "sampling/importance_sampling_ratio/min": 0.5066708326339722, + "sampling/importance_sampling_ratio/mean": 0.9474743604660034, + "sampling/importance_sampling_ratio/max": 1.406033992767334, + "kl": 8.065076605134891e-05, + "entropy": 0.026434177794726565, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.46844732761383, + "epoch": 5.859375e-05, + "step": 3 + }, + { + "loss": -0.4021121561527252, + "grad_norm": 1.4497005939483643, + "learning_rate": 3e-07, + "num_tokens": 43876.0, + "completions/mean_length": 745.0, + "completions/min_length": 203.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 455.3333435058594, + "completions/min_terminated_length": 203.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 17.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.012500000186264515, + "rewards/reward_func/std": 0.06943651288747787, + "reward": 0.012500000186264515, + "reward_std": 0.06943650543689728, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017801430076360703, + "sampling/sampling_logp_difference/max": 0.514519453048706, + "sampling/importance_sampling_ratio/min": 0.7789422273635864, + "sampling/importance_sampling_ratio/mean": 0.989887535572052, + "sampling/importance_sampling_ratio/max": 1.3373364210128784, + "kl": 0.00016320715712936362, + "entropy": 0.02744299836922437, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.301488742232323, + "epoch": 7.8125e-05, + "step": 4 + }, + { + "loss": -0.24790050089359283, + "grad_norm": 2.026782989501953, + "learning_rate": 4e-07, + "num_tokens": 55224.0, + "completions/mean_length": 733.875, + "completions/min_length": 178.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 451.0, + "completions/min_terminated_length": 178.0, + "completions/max_terminated_length": 901.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.012500000186264515, + "rewards/reward_func/std": 0.06943651288747787, + "reward": 0.012500000186264515, + "reward_std": 0.06943650543689728, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0012523536570370197, + "sampling/sampling_logp_difference/max": 0.8087813854217529, + "sampling/importance_sampling_ratio/min": 0.3716883957386017, + "sampling/importance_sampling_ratio/mean": 0.8639620542526245, + "sampling/importance_sampling_ratio/max": 1.6881294250488281, + "kl": 0.00013307032543252717, + "entropy": 0.02254073432413861, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.515335952863097, + "epoch": 9.765625e-05, + "step": 5 + }, + { + "loss": -0.11606968194246292, + "grad_norm": 1.2161829471588135, + "learning_rate": 5e-07, + "num_tokens": 66442.0, + "completions/mean_length": 717.75, + "completions/min_length": 145.0, + "completions/max_length": 935.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 397.66668701171875, + "completions/min_terminated_length": 145.0, + "completions/max_terminated_length": 879.0, + "tools/call_frequency": 17.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.09687500447034836, + "rewards/reward_func/std": 0.2626912593841553, + "reward": 0.09687500447034836, + "reward_std": 0.2626912593841553, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001172104268334806, + "sampling/sampling_logp_difference/max": 0.4760704040527344, + "sampling/importance_sampling_ratio/min": 0.3320813477039337, + "sampling/importance_sampling_ratio/mean": 0.96290123462677, + "sampling/importance_sampling_ratio/max": 1.736915946006775, + "kl": 5.124416792057218e-05, + "entropy": 0.02150688081746921, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.084405278787017, + "epoch": 0.0001171875, + "step": 6 + }, + { + "loss": 0.43859559297561646, + "grad_norm": 1.5172673463821411, + "learning_rate": 6e-07, + "num_tokens": 78054.0, + "completions/mean_length": 765.625, + "completions/min_length": 257.0, + "completions/max_length": 1003.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 598.75, + "completions/min_terminated_length": 257.0, + "completions/max_terminated_length": 908.0, + "tools/call_frequency": 18.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.140625, + "rewards/reward_func/std": 0.1689978390932083, + "reward": 0.140625, + "reward_std": 0.1689978390932083, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018408597679808736, + "sampling/sampling_logp_difference/max": 1.9139561653137207, + "sampling/importance_sampling_ratio/min": 0.2365708351135254, + "sampling/importance_sampling_ratio/mean": 0.9503114223480225, + "sampling/importance_sampling_ratio/max": 2.3263304233551025, + "kl": 0.000316470254745127, + "entropy": 0.03178998743533157, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.839783350005746, + "epoch": 0.00013671875, + "step": 7 + }, + { + "loss": -0.6623136401176453, + "grad_norm": 2.9550883769989014, + "learning_rate": 7e-07, + "num_tokens": 88877.0, + "completions/mean_length": 667.125, + "completions/min_length": 214.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 257.66668701171875, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 345.0, + "tools/call_frequency": 16.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.012500000186264515, + "rewards/reward_func/std": 0.06943651288747787, + "reward": 0.012500000186264515, + "reward_std": 0.06943650543689728, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018735796911641955, + "sampling/sampling_logp_difference/max": 0.5645437240600586, + "sampling/importance_sampling_ratio/min": 0.3759009838104248, + "sampling/importance_sampling_ratio/mean": 1.309678316116333, + "sampling/importance_sampling_ratio/max": 2.1810364723205566, + "kl": 0.0003264702955299015, + "entropy": 0.038075507909525186, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.00939847715199, + "epoch": 0.00015625, + "step": 8 + }, + { + "loss": -0.060755811631679535, + "grad_norm": 2.465081214904785, + "learning_rate": 8e-07, + "num_tokens": 99022.0, + "completions/mean_length": 582.25, + "completions/min_length": 140.0, + "completions/max_length": 1003.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 350.8000183105469, + "completions/min_terminated_length": 140.0, + "completions/max_terminated_length": 906.0, + "tools/call_frequency": 13.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.04218750074505806, + "rewards/reward_func/std": 0.1761362999677658, + "reward": 0.04218750074505806, + "reward_std": 0.1761362999677658, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021804259158670902, + "sampling/sampling_logp_difference/max": 0.43540024757385254, + "sampling/importance_sampling_ratio/min": 0.4773575961589813, + "sampling/importance_sampling_ratio/mean": 0.9053957462310791, + "sampling/importance_sampling_ratio/max": 1.5945932865142822, + "kl": 0.0002964280129162944, + "entropy": 0.036511566722765565, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.046199945732951, + "epoch": 0.00017578125, + "step": 9 + }, + { + "loss": -0.3869200050830841, + "grad_norm": 2.3748059272766113, + "learning_rate": 9e-07, + "num_tokens": 110381.0, + "completions/mean_length": 734.75, + "completions/min_length": 168.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 623.4000244140625, + "completions/min_terminated_length": 168.0, + "completions/max_terminated_length": 940.0, + "tools/call_frequency": 16.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.012500000186264515, + "rewards/reward_func/std": 0.06943651288747787, + "reward": 0.012500000186264515, + "reward_std": 0.06943650543689728, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0015139196766540408, + "sampling/sampling_logp_difference/max": 0.6627935171127319, + "sampling/importance_sampling_ratio/min": 0.4610547721385956, + "sampling/importance_sampling_ratio/mean": 1.08158540725708, + "sampling/importance_sampling_ratio/max": 1.6864471435546875, + "kl": 0.000204855328775011, + "entropy": 0.02492033498128876, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.244448188692331, + "epoch": 0.0001953125, + "step": 10 + }, + { + "loss": 0.3022535443305969, + "grad_norm": 1.7455980777740479, + "learning_rate": 1e-06, + "num_tokens": 121220.0, + "completions/mean_length": 668.625, + "completions/min_length": 154.0, + "completions/max_length": 977.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 233.6666717529297, + "completions/min_terminated_length": 154.0, + "completions/max_terminated_length": 310.0, + "tools/call_frequency": 15.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.15781250596046448, + "rewards/reward_func/std": 0.31938400864601135, + "reward": 0.15781250596046448, + "reward_std": 0.31938400864601135, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002131029265001416, + "sampling/sampling_logp_difference/max": 0.6339419484138489, + "sampling/importance_sampling_ratio/min": 0.4307011365890503, + "sampling/importance_sampling_ratio/mean": 0.972503662109375, + "sampling/importance_sampling_ratio/max": 2.711538076400757, + "kl": 0.00035725860936963727, + "entropy": 0.052559162373654544, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.53186353482306, + "epoch": 0.00021484375, + "step": 11 + }, + { + "loss": 0.3819582164287567, + "grad_norm": 1.74873948097229, + "learning_rate": 9.974358974358974e-07, + "num_tokens": 130636.0, + "completions/mean_length": 490.625, + "completions/min_length": 139.0, + "completions/max_length": 944.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 344.5, + "completions/min_terminated_length": 139.0, + "completions/max_terminated_length": 944.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.15850694477558136, + "rewards/reward_func/std": 0.30659136176109314, + "reward": 0.15850694477558136, + "reward_std": 0.30659136176109314, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028385394252836704, + "sampling/sampling_logp_difference/max": 0.6382479667663574, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9048388004302979, + "sampling/importance_sampling_ratio/max": 1.8268561363220215, + "kl": 0.0003670242924727063, + "entropy": 0.05921977257821709, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.151850877329707, + "epoch": 0.000234375, + "step": 12 + }, + { + "loss": 0.09304837137460709, + "grad_norm": 2.0897233486175537, + "learning_rate": 9.948717948717949e-07, + "num_tokens": 141428.0, + "completions/mean_length": 663.875, + "completions/min_length": 178.0, + "completions/max_length": 945.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 503.0, + "completions/min_terminated_length": 178.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.09479166567325592, + "rewards/reward_func/std": 0.2570026218891144, + "reward": 0.09479166567325592, + "reward_std": 0.2570026218891144, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018424922600388527, + "sampling/sampling_logp_difference/max": 0.3389971852302551, + "sampling/importance_sampling_ratio/min": 0.5393957495689392, + "sampling/importance_sampling_ratio/mean": 0.8943148255348206, + "sampling/importance_sampling_ratio/max": 1.3376030921936035, + "kl": 0.00034260294887644704, + "entropy": 0.040203147334977984, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.233821969479322, + "epoch": 0.00025390625, + "step": 13 + }, + { + "loss": -0.11376293003559113, + "grad_norm": 3.1292147636413574, + "learning_rate": 9.923076923076923e-07, + "num_tokens": 151477.0, + "completions/mean_length": 568.875, + "completions/min_length": 167.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 457.0, + "completions/min_terminated_length": 167.0, + "completions/max_terminated_length": 933.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.11292614042758942, + "rewards/reward_func/std": 0.21946066617965698, + "reward": 0.11292614042758942, + "reward_std": 0.21946066617965698, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024365053977817297, + "sampling/sampling_logp_difference/max": 1.144789218902588, + "sampling/importance_sampling_ratio/min": 0.611109733581543, + "sampling/importance_sampling_ratio/mean": 1.346110463142395, + "sampling/importance_sampling_ratio/max": 2.8036210536956787, + "kl": 0.00042028062853205483, + "entropy": 0.03531385416863486, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.074251499027014, + "epoch": 0.0002734375, + "step": 14 + }, + { + "loss": 0.21044254302978516, + "grad_norm": 2.2925565242767334, + "learning_rate": 9.897435897435898e-07, + "num_tokens": 161599.0, + "completions/mean_length": 580.0, + "completions/min_length": 197.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 385.20001220703125, + "completions/min_terminated_length": 197.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.12604166567325592, + "rewards/reward_func/std": 0.2134113311767578, + "reward": 0.12604166567325592, + "reward_std": 0.2134113311767578, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0022266784217208624, + "sampling/sampling_logp_difference/max": 0.6686651706695557, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9781344532966614, + "sampling/importance_sampling_ratio/max": 1.4612537622451782, + "kl": 0.0005073034571978496, + "entropy": 0.05668710730969906, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.329704467207193, + "epoch": 0.00029296875, + "step": 15 + }, + { + "loss": 0.29178959131240845, + "grad_norm": 1.1052141189575195, + "learning_rate": 9.871794871794872e-07, + "num_tokens": 173171.0, + "completions/mean_length": 761.5, + "completions/min_length": 273.0, + "completions/max_length": 972.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 511.3333435058594, + "completions/min_terminated_length": 273.0, + "completions/max_terminated_length": 972.0, + "tools/call_frequency": 17.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.15156251192092896, + "rewards/reward_func/std": 0.18832409381866455, + "reward": 0.15156251192092896, + "reward_std": 0.18832407891750336, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0015812081983312964, + "sampling/sampling_logp_difference/max": 0.5208761692047119, + "sampling/importance_sampling_ratio/min": 0.5423237681388855, + "sampling/importance_sampling_ratio/mean": 0.732074499130249, + "sampling/importance_sampling_ratio/max": 0.9245461821556091, + "kl": 0.00031370692192922434, + "entropy": 0.03615973691921681, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.792964367195964, + "epoch": 0.0003125, + "step": 16 + }, + { + "loss": -0.022374901920557022, + "grad_norm": 1.346390962600708, + "learning_rate": 9.846153846153847e-07, + "num_tokens": 185270.0, + "completions/mean_length": 826.875, + "completions/min_length": 260.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 758.25, + "completions/min_terminated_length": 260.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 20.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.08437500149011612, + "rewards/reward_func/std": 0.16633524000644684, + "reward": 0.08437500149011612, + "reward_std": 0.16633524000644684, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0008454358321614563, + "sampling/sampling_logp_difference/max": 0.4660501480102539, + "sampling/importance_sampling_ratio/min": 0.5168011784553528, + "sampling/importance_sampling_ratio/mean": 1.059901475906372, + "sampling/importance_sampling_ratio/max": 2.511544704437256, + "kl": 0.0002804777855089924, + "entropy": 0.01317918678978458, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.240736737847328, + "epoch": 0.00033203125, + "step": 17 + }, + { + "loss": 0.45776602625846863, + "grad_norm": 3.0103297233581543, + "learning_rate": 9.820512820512819e-07, + "num_tokens": 197425.0, + "completions/mean_length": 834.125, + "completions/min_length": 277.0, + "completions/max_length": 982.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 751.75, + "completions/min_terminated_length": 277.0, + "completions/max_terminated_length": 982.0, + "tools/call_frequency": 19.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.27812498807907104, + "rewards/reward_func/std": 0.28729504346847534, + "reward": 0.27812498807907104, + "reward_std": 0.28729504346847534, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018759757513180375, + "sampling/sampling_logp_difference/max": 0.9557504653930664, + "sampling/importance_sampling_ratio/min": 0.2504982352256775, + "sampling/importance_sampling_ratio/mean": 1.0488845109939575, + "sampling/importance_sampling_ratio/max": 2.066556453704834, + "kl": 0.0008316919943354151, + "entropy": 0.027564861753489822, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.271112009882927, + "epoch": 0.0003515625, + "step": 18 + }, + { + "loss": 0.01822827383875847, + "grad_norm": 3.132244110107422, + "learning_rate": 9.794871794871793e-07, + "num_tokens": 207574.0, + "completions/mean_length": 583.875, + "completions/min_length": 227.0, + "completions/max_length": 920.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 388.8000183105469, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 902.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.1979166567325592, + "rewards/reward_func/std": 0.373575359582901, + "reward": 0.1979166567325592, + "reward_std": 0.373575359582901, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002337060170248151, + "sampling/sampling_logp_difference/max": 0.6929893493652344, + "sampling/importance_sampling_ratio/min": 0.495841383934021, + "sampling/importance_sampling_ratio/mean": 0.9285518527030945, + "sampling/importance_sampling_ratio/max": 2.1295344829559326, + "kl": 0.001089137746475899, + "entropy": 0.05025588144781068, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.784936871379614, + "epoch": 0.00037109375, + "step": 19 + }, + { + "loss": 0.18115338683128357, + "grad_norm": 1.4740865230560303, + "learning_rate": 9.769230769230768e-07, + "num_tokens": 219032.0, + "completions/mean_length": 747.125, + "completions/min_length": 235.0, + "completions/max_length": 958.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 586.0, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 916.0, + "tools/call_frequency": 17.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.2541666626930237, + "rewards/reward_func/std": 0.29371729493141174, + "reward": 0.2541666626930237, + "reward_std": 0.29371729493141174, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018756315112113953, + "sampling/sampling_logp_difference/max": 0.648700475692749, + "sampling/importance_sampling_ratio/min": 0.23824924230575562, + "sampling/importance_sampling_ratio/mean": 0.7365133762359619, + "sampling/importance_sampling_ratio/max": 1.3203972578048706, + "kl": 0.0015302523515856592, + "entropy": 0.03008002700516954, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.007541421800852, + "epoch": 0.000390625, + "step": 20 + }, + { + "loss": -0.06809201836585999, + "grad_norm": 1.8541380167007446, + "learning_rate": 9.743589743589742e-07, + "num_tokens": 230312.0, + "completions/mean_length": 723.75, + "completions/min_length": 253.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 453.3333435058594, + "completions/min_terminated_length": 253.0, + "completions/max_terminated_length": 819.0, + "tools/call_frequency": 18.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3421874940395355, + "rewards/reward_func/std": 0.356657475233078, + "reward": 0.3421874940395355, + "reward_std": 0.356657475233078, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002288880292326212, + "sampling/sampling_logp_difference/max": 1.1105562448501587, + "sampling/importance_sampling_ratio/min": 0.08672235906124115, + "sampling/importance_sampling_ratio/mean": 0.8479963541030884, + "sampling/importance_sampling_ratio/max": 2.3816473484039307, + "kl": 0.0016434401059086667, + "entropy": 0.03622693271609023, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.05686098150909, + "epoch": 0.00041015625, + "step": 21 + }, + { + "loss": -0.07511341571807861, + "grad_norm": 1.3750392198562622, + "learning_rate": 9.717948717948717e-07, + "num_tokens": 241816.0, + "completions/mean_length": 752.875, + "completions/min_length": 282.0, + "completions/max_length": 981.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 585.25, + "completions/min_terminated_length": 282.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 17.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.434027761220932, + "rewards/reward_func/std": 0.25980231165885925, + "reward": 0.434027761220932, + "reward_std": 0.25980231165885925, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019160519586876035, + "sampling/sampling_logp_difference/max": 0.9354541301727295, + "sampling/importance_sampling_ratio/min": 0.3013438582420349, + "sampling/importance_sampling_ratio/mean": 0.8698049783706665, + "sampling/importance_sampling_ratio/max": 1.3207614421844482, + "kl": 0.0023370416420220863, + "entropy": 0.037189144699368626, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.464417094364762, + "epoch": 0.0004296875, + "step": 22 + }, + { + "loss": -0.06243608146905899, + "grad_norm": 1.7572788000106812, + "learning_rate": 9.692307692307691e-07, + "num_tokens": 253310.0, + "completions/mean_length": 751.75, + "completions/min_length": 299.0, + "completions/max_length": 967.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 495.0, + "completions/min_terminated_length": 299.0, + "completions/max_terminated_length": 879.0, + "tools/call_frequency": 17.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3775094747543335, + "rewards/reward_func/std": 0.287571519613266, + "reward": 0.3775094747543335, + "reward_std": 0.287571519613266, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018008759943768382, + "sampling/sampling_logp_difference/max": 0.6085361838340759, + "sampling/importance_sampling_ratio/min": 0.5529455542564392, + "sampling/importance_sampling_ratio/mean": 1.1744285821914673, + "sampling/importance_sampling_ratio/max": 2.702420473098755, + "kl": 0.0022118995038908906, + "entropy": 0.031101019820198417, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.165682626888156, + "epoch": 0.00044921875, + "step": 23 + }, + { + "loss": 0.8031491637229919, + "grad_norm": 4.549752235412598, + "learning_rate": 9.666666666666666e-07, + "num_tokens": 263606.0, + "completions/mean_length": 601.0, + "completions/min_length": 179.0, + "completions/max_length": 980.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 270.75, + "completions/min_terminated_length": 179.0, + "completions/max_terminated_length": 347.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.29826387763023376, + "rewards/reward_func/std": 0.23498418927192688, + "reward": 0.29826387763023376, + "reward_std": 0.23498417437076569, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0027140614110976458, + "sampling/sampling_logp_difference/max": 0.6200490593910217, + "sampling/importance_sampling_ratio/min": 0.4262768030166626, + "sampling/importance_sampling_ratio/mean": 1.1970113515853882, + "sampling/importance_sampling_ratio/max": 2.80903959274292, + "kl": 0.004755028929139371, + "entropy": 0.04784228530479595, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.914157379418612, + "epoch": 0.00046875, + "step": 24 + }, + { + "loss": -0.27596211433410645, + "grad_norm": 2.18823504447937, + "learning_rate": 9.64102564102564e-07, + "num_tokens": 275560.0, + "completions/mean_length": 809.625, + "completions/min_length": 290.0, + "completions/max_length": 922.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 575.0, + "completions/min_terminated_length": 290.0, + "completions/max_terminated_length": 860.0, + "tools/call_frequency": 19.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.36584821343421936, + "rewards/reward_func/std": 0.3467599153518677, + "reward": 0.36584821343421936, + "reward_std": 0.3467599153518677, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0014124035369604826, + "sampling/sampling_logp_difference/max": 0.5645416975021362, + "sampling/importance_sampling_ratio/min": 0.5617415308952332, + "sampling/importance_sampling_ratio/mean": 0.947551965713501, + "sampling/importance_sampling_ratio/max": 1.8321820497512817, + "kl": 0.002324921191757312, + "entropy": 0.022645775898126885, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.093006016686559, + "epoch": 0.00048828125, + "step": 25 + }, + { + "loss": 0.4539790153503418, + "grad_norm": 2.1662776470184326, + "learning_rate": 9.615384615384615e-07, + "num_tokens": 286979.0, + "completions/mean_length": 742.5, + "completions/min_length": 179.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 640.0, + "completions/min_terminated_length": 179.0, + "completions/max_terminated_length": 938.0, + "tools/call_frequency": 16.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.41843751072883606, + "rewards/reward_func/std": 0.3856533169746399, + "reward": 0.41843751072883606, + "reward_std": 0.3856533169746399, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020094425417482853, + "sampling/sampling_logp_difference/max": 0.4366832971572876, + "sampling/importance_sampling_ratio/min": 0.5928208231925964, + "sampling/importance_sampling_ratio/mean": 1.0782763957977295, + "sampling/importance_sampling_ratio/max": 2.0687148571014404, + "kl": 0.003962378337746486, + "entropy": 0.03329420986119658, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.880603231489658, + "epoch": 0.0005078125, + "step": 26 + }, + { + "loss": -0.06831827759742737, + "grad_norm": 2.242339849472046, + "learning_rate": 9.58974358974359e-07, + "num_tokens": 298392.0, + "completions/mean_length": 741.125, + "completions/min_length": 283.0, + "completions/max_length": 971.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 657.2000122070312, + "completions/min_terminated_length": 283.0, + "completions/max_terminated_length": 971.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5, + "rewards/reward_func/std": 0.29813408851623535, + "reward": 0.5, + "reward_std": 0.29813408851623535, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001771795330569148, + "sampling/sampling_logp_difference/max": 1.0282599925994873, + "sampling/importance_sampling_ratio/min": 0.31944358348846436, + "sampling/importance_sampling_ratio/mean": 0.9405813217163086, + "sampling/importance_sampling_ratio/max": 1.1779206991195679, + "kl": 0.0048524205485591665, + "entropy": 0.03593305914546363, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.951548459008336, + "epoch": 0.00052734375, + "step": 27 + }, + { + "loss": -0.19446192681789398, + "grad_norm": 1.1434029340744019, + "learning_rate": 9.564102564102564e-07, + "num_tokens": 310244.0, + "completions/mean_length": 796.75, + "completions/min_length": 283.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 724.0, + "completions/min_terminated_length": 283.0, + "completions/max_terminated_length": 897.0, + "tools/call_frequency": 19.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5276041626930237, + "rewards/reward_func/std": 0.19628164172172546, + "reward": 0.5276041626930237, + "reward_std": 0.19628162682056427, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013002228224650025, + "sampling/sampling_logp_difference/max": 0.6832635402679443, + "sampling/importance_sampling_ratio/min": 0.5803820490837097, + "sampling/importance_sampling_ratio/mean": 0.9427469968795776, + "sampling/importance_sampling_ratio/max": 1.3937605619430542, + "kl": 0.005832243448821828, + "entropy": 0.02038799930596724, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.477646362036467, + "epoch": 0.000546875, + "step": 28 + }, + { + "loss": -0.05061045289039612, + "grad_norm": 1.4409914016723633, + "learning_rate": 9.538461538461538e-07, + "num_tokens": 322217.0, + "completions/mean_length": 812.5, + "completions/min_length": 226.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 739.75, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 954.0, + "tools/call_frequency": 18.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35538196563720703, + "rewards/reward_func/std": 0.28778064250946045, + "reward": 0.35538196563720703, + "reward_std": 0.28778064250946045, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011515418300405145, + "sampling/sampling_logp_difference/max": 0.6983804702758789, + "sampling/importance_sampling_ratio/min": 0.40979722142219543, + "sampling/importance_sampling_ratio/mean": 0.9165198802947998, + "sampling/importance_sampling_ratio/max": 1.4086145162582397, + "kl": 0.003760237668757327, + "entropy": 0.016279009345453233, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.393253592774272, + "epoch": 0.00056640625, + "step": 29 + }, + { + "loss": -0.385548859834671, + "grad_norm": 1.7927696704864502, + "learning_rate": 9.512820512820512e-07, + "num_tokens": 332796.0, + "completions/mean_length": 637.125, + "completions/min_length": 172.0, + "completions/max_length": 899.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 228.33334350585938, + "completions/min_terminated_length": 172.0, + "completions/max_terminated_length": 295.0, + "tools/call_frequency": 15.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5516319274902344, + "rewards/reward_func/std": 0.32275959849357605, + "reward": 0.5516319274902344, + "reward_std": 0.32275959849357605, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011996538378298283, + "sampling/sampling_logp_difference/max": 0.6487948894500732, + "sampling/importance_sampling_ratio/min": 0.46611258387565613, + "sampling/importance_sampling_ratio/mean": 1.055126667022705, + "sampling/importance_sampling_ratio/max": 1.4563429355621338, + "kl": 0.008034612430492416, + "entropy": 0.02835029154084623, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.59170452132821, + "epoch": 0.0005859375, + "step": 30 + }, + { + "loss": 0.037692248821258545, + "grad_norm": 3.0362234115600586, + "learning_rate": 9.487179487179486e-07, + "num_tokens": 344016.0, + "completions/mean_length": 717.0, + "completions/min_length": 225.0, + "completions/max_length": 904.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 554.75, + "completions/min_terminated_length": 225.0, + "completions/max_terminated_length": 877.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47845643758773804, + "rewards/reward_func/std": 0.2983555495738983, + "reward": 0.47845643758773804, + "reward_std": 0.2983555197715759, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001593840541318059, + "sampling/sampling_logp_difference/max": 0.4523376226425171, + "sampling/importance_sampling_ratio/min": 0.45376721024513245, + "sampling/importance_sampling_ratio/mean": 0.9045858979225159, + "sampling/importance_sampling_ratio/max": 1.4367549419403076, + "kl": 0.010412814284791239, + "entropy": 0.028418905043508857, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.265132617205381, + "epoch": 0.00060546875, + "step": 31 + }, + { + "loss": 0.07955804467201233, + "grad_norm": 1.6195743083953857, + "learning_rate": 9.461538461538461e-07, + "num_tokens": 355955.0, + "completions/mean_length": 808.375, + "completions/min_length": 294.0, + "completions/max_length": 918.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 760.2000122070312, + "completions/min_terminated_length": 294.0, + "completions/max_terminated_length": 918.0, + "tools/call_frequency": 18.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5595486164093018, + "rewards/reward_func/std": 0.24332065880298615, + "reward": 0.5595486164093018, + "reward_std": 0.24332064390182495, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0014648770447820425, + "sampling/sampling_logp_difference/max": 0.49565935134887695, + "sampling/importance_sampling_ratio/min": 0.6601452231407166, + "sampling/importance_sampling_ratio/mean": 0.9359903335571289, + "sampling/importance_sampling_ratio/max": 1.41331148147583, + "kl": 0.008377619262319058, + "entropy": 0.025353577395435423, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.834929330274463, + "epoch": 0.000625, + "step": 32 + }, + { + "loss": -0.006198972463607788, + "grad_norm": 1.172409176826477, + "learning_rate": 9.435897435897435e-07, + "num_tokens": 368478.0, + "completions/mean_length": 880.375, + "completions/min_length": 820.0, + "completions/max_length": 910.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 894.0, + "completions/min_terminated_length": 880.0, + "completions/max_terminated_length": 910.0, + "tools/call_frequency": 21.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5701388716697693, + "rewards/reward_func/std": 0.23876658082008362, + "reward": 0.5701388716697693, + "reward_std": 0.23876658082008362, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0009140048641711473, + "sampling/sampling_logp_difference/max": 0.596961498260498, + "sampling/importance_sampling_ratio/min": 0.6474210023880005, + "sampling/importance_sampling_ratio/mean": 0.9551972150802612, + "sampling/importance_sampling_ratio/max": 1.4585293531417847, + "kl": 0.0053108767606318, + "entropy": 0.009245342953363433, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.6329645216465, + "epoch": 0.00064453125, + "step": 33 + }, + { + "loss": -0.2050226926803589, + "grad_norm": 0.9721785187721252, + "learning_rate": 9.41025641025641e-07, + "num_tokens": 381760.0, + "completions/mean_length": 974.625, + "completions/min_length": 905.0, + "completions/max_length": 1016.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 960.5, + "completions/min_terminated_length": 905.0, + "completions/max_terminated_length": 1016.0, + "tools/call_frequency": 20.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.08124999701976776, + "rewards/reward_func/std": 0.015526476316154003, + "reward": 0.08124999701976776, + "reward_std": 0.015526475384831429, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0007203315617516637, + "sampling/sampling_logp_difference/max": 0.3977140188217163, + "sampling/importance_sampling_ratio/min": 0.6550352573394775, + "sampling/importance_sampling_ratio/mean": 1.053494930267334, + "sampling/importance_sampling_ratio/max": 1.9253982305526733, + "kl": 0.0008235033310484141, + "entropy": 0.00919160939520225, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.283352492377162, + "epoch": 0.0006640625, + "step": 34 + }, + { + "loss": 0.22937515377998352, + "grad_norm": 1.7631758451461792, + "learning_rate": 9.384615384615384e-07, + "num_tokens": 393586.0, + "completions/mean_length": 794.375, + "completions/min_length": 184.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 523.0, + "completions/min_terminated_length": 184.0, + "completions/max_terminated_length": 862.0, + "tools/call_frequency": 18.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6572916507720947, + "rewards/reward_func/std": 0.22179989516735077, + "reward": 0.6572916507720947, + "reward_std": 0.22179989516735077, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0009160267654806376, + "sampling/sampling_logp_difference/max": 0.339185893535614, + "sampling/importance_sampling_ratio/min": 0.6892018914222717, + "sampling/importance_sampling_ratio/mean": 1.0936065912246704, + "sampling/importance_sampling_ratio/max": 1.8527777194976807, + "kl": 0.00752565820585005, + "entropy": 0.017163428070489317, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.158073622733355, + "epoch": 0.00068359375, + "step": 35 + }, + { + "loss": -0.28859877586364746, + "grad_norm": 7.170706272125244, + "learning_rate": 9.358974358974359e-07, + "num_tokens": 403734.0, + "completions/mean_length": 582.5, + "completions/min_length": 234.0, + "completions/max_length": 939.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 539.857177734375, + "completions/min_terminated_length": 234.0, + "completions/max_terminated_length": 939.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6010416746139526, + "rewards/reward_func/std": 0.2227705866098404, + "reward": 0.6010416746139526, + "reward_std": 0.2227706015110016, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00295042316429317, + "sampling/sampling_logp_difference/max": 1.6750982999801636, + "sampling/importance_sampling_ratio/min": 0.09173579514026642, + "sampling/importance_sampling_ratio/mean": 0.8933800458908081, + "sampling/importance_sampling_ratio/max": 1.5766624212265015, + "kl": 0.010616067243972793, + "entropy": 0.058151982026174664, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.160806702449918, + "epoch": 0.000703125, + "step": 36 + }, + { + "loss": -0.2517475187778473, + "grad_norm": 2.3694074153900146, + "learning_rate": 9.333333333333333e-07, + "num_tokens": 414605.0, + "completions/mean_length": 672.625, + "completions/min_length": 314.0, + "completions/max_length": 900.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 558.7999877929688, + "completions/min_terminated_length": 314.0, + "completions/max_terminated_length": 900.0, + "tools/call_frequency": 15.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5253472328186035, + "rewards/reward_func/std": 0.2689405381679535, + "reward": 0.5253472328186035, + "reward_std": 0.2689405381679535, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002966368105262518, + "sampling/sampling_logp_difference/max": 0.4760727882385254, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.8815990686416626, + "sampling/importance_sampling_ratio/max": 1.423635482788086, + "kl": 0.007583088823594153, + "entropy": 0.052448477305006236, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.982416234910488, + "epoch": 0.00072265625, + "step": 37 + }, + { + "loss": -0.018093064427375793, + "grad_norm": 4.158572673797607, + "learning_rate": 9.307692307692308e-07, + "num_tokens": 425285.0, + "completions/mean_length": 650.375, + "completions/min_length": 238.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 419.75, + "completions/min_terminated_length": 238.0, + "completions/max_terminated_length": 843.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6832228899002075, + "rewards/reward_func/std": 0.1575513631105423, + "reward": 0.6832228899002075, + "reward_std": 0.1575513482093811, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019418722949922085, + "sampling/sampling_logp_difference/max": 0.3286604881286621, + "sampling/importance_sampling_ratio/min": 0.3631701171398163, + "sampling/importance_sampling_ratio/mean": 1.2346949577331543, + "sampling/importance_sampling_ratio/max": 2.448169708251953, + "kl": 0.009506175760179758, + "entropy": 0.04159332701237872, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.742824003100395, + "epoch": 0.0007421875, + "step": 38 + }, + { + "loss": -0.07837707549333572, + "grad_norm": 2.4153051376342773, + "learning_rate": 9.282051282051282e-07, + "num_tokens": 435352.0, + "completions/mean_length": 574.0, + "completions/min_length": 216.0, + "completions/max_length": 892.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 397.20001220703125, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 892.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.704265832901001, + "rewards/reward_func/std": 0.18740791082382202, + "reward": 0.704265832901001, + "reward_std": 0.18740791082382202, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001970284851267934, + "sampling/sampling_logp_difference/max": 0.4532172679901123, + "sampling/importance_sampling_ratio/min": 0.7691821455955505, + "sampling/importance_sampling_ratio/mean": 1.0129979848861694, + "sampling/importance_sampling_ratio/max": 1.3600549697875977, + "kl": 0.009227094182278961, + "entropy": 0.04619215623824857, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.025273127481341, + "epoch": 0.00076171875, + "step": 39 + }, + { + "loss": -0.061302538961172104, + "grad_norm": 1.14959716796875, + "learning_rate": 9.256410256410257e-07, + "num_tokens": 447313.0, + "completions/mean_length": 808.875, + "completions/min_length": 270.0, + "completions/max_length": 935.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 719.5, + "completions/min_terminated_length": 270.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 18.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6382936239242554, + "rewards/reward_func/std": 0.2232269048690796, + "reward": 0.6382936239242554, + "reward_std": 0.2232269048690796, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0012364190770313144, + "sampling/sampling_logp_difference/max": 0.557631254196167, + "sampling/importance_sampling_ratio/min": 0.375461608171463, + "sampling/importance_sampling_ratio/mean": 0.8361858129501343, + "sampling/importance_sampling_ratio/max": 1.5748335123062134, + "kl": 0.004282362875528634, + "entropy": 0.020274960552342236, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.238981368020177, + "epoch": 0.00078125, + "step": 40 + }, + { + "loss": -0.18477532267570496, + "grad_norm": 3.5714755058288574, + "learning_rate": 9.230769230769231e-07, + "num_tokens": 458171.0, + "completions/mean_length": 671.375, + "completions/min_length": 262.0, + "completions/max_length": 944.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 533.7999877929688, + "completions/min_terminated_length": 262.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 16.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6512153148651123, + "rewards/reward_func/std": 0.23627381026744843, + "reward": 0.6512153148651123, + "reward_std": 0.23627381026744843, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001986459130421281, + "sampling/sampling_logp_difference/max": 0.7514872550964355, + "sampling/importance_sampling_ratio/min": 0.5959795117378235, + "sampling/importance_sampling_ratio/mean": 1.2724785804748535, + "sampling/importance_sampling_ratio/max": 2.3634707927703857, + "kl": 0.005765270347183105, + "entropy": 0.03929175069788471, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.785259056836367, + "epoch": 0.00080078125, + "step": 41 + }, + { + "loss": 0.11729055643081665, + "grad_norm": 1.7894781827926636, + "learning_rate": 9.205128205128205e-07, + "num_tokens": 468365.0, + "completions/mean_length": 588.25, + "completions/min_length": 246.0, + "completions/max_length": 951.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 261.5, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 271.0, + "tools/call_frequency": 13.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5203661918640137, + "rewards/reward_func/std": 0.21326640248298645, + "reward": 0.5203661918640137, + "reward_std": 0.21326638758182526, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028968574479222298, + "sampling/sampling_logp_difference/max": 1.675093173980713, + "sampling/importance_sampling_ratio/min": 0.698772668838501, + "sampling/importance_sampling_ratio/mean": 1.0332955121994019, + "sampling/importance_sampling_ratio/max": 2.200169801712036, + "kl": 0.011474523867946118, + "entropy": 0.05125397277879529, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.707072261720896, + "epoch": 0.0008203125, + "step": 42 + }, + { + "loss": -0.3338780105113983, + "grad_norm": 1.0523158311843872, + "learning_rate": 9.179487179487179e-07, + "num_tokens": 478314.0, + "completions/mean_length": 558.125, + "completions/min_length": 75.0, + "completions/max_length": 895.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 366.8000183105469, + "completions/min_terminated_length": 75.0, + "completions/max_terminated_length": 886.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5324652194976807, + "rewards/reward_func/std": 0.27418607473373413, + "reward": 0.5324652194976807, + "reward_std": 0.27418604493141174, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020767918322235346, + "sampling/sampling_logp_difference/max": 1.0686956644058228, + "sampling/importance_sampling_ratio/min": 0.2878314256668091, + "sampling/importance_sampling_ratio/mean": 0.7563162446022034, + "sampling/importance_sampling_ratio/max": 1.160767674446106, + "kl": 0.00523930269991979, + "entropy": 0.045632984925759956, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.356886321678758, + "epoch": 0.00083984375, + "step": 43 + }, + { + "loss": 0.2786228060722351, + "grad_norm": 2.7462081909179688, + "learning_rate": 9.153846153846153e-07, + "num_tokens": 487794.0, + "completions/mean_length": 499.5, + "completions/min_length": 221.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 499.5, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 929.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5947916507720947, + "rewards/reward_func/std": 0.20592357218265533, + "reward": 0.5947916507720947, + "reward_std": 0.20592355728149414, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003328504506498575, + "sampling/sampling_logp_difference/max": 0.46151202917099, + "sampling/importance_sampling_ratio/min": 0.31267663836479187, + "sampling/importance_sampling_ratio/mean": 0.9779818058013916, + "sampling/importance_sampling_ratio/max": 1.4097931385040283, + "kl": 0.0067671629367396235, + "entropy": 0.06303813750855625, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.21711173094809, + "epoch": 0.000859375, + "step": 44 + }, + { + "loss": 0.010867382399737835, + "grad_norm": 2.7273166179656982, + "learning_rate": 9.128205128205127e-07, + "num_tokens": 496688.0, + "completions/mean_length": 426.5, + "completions/min_length": 236.0, + "completions/max_length": 866.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 363.71429443359375, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 853.0, + "tools/call_frequency": 10.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5804356336593628, + "rewards/reward_func/std": 0.22475191950798035, + "reward": 0.5804356336593628, + "reward_std": 0.22475191950798035, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004695659503340721, + "sampling/sampling_logp_difference/max": 0.7453341484069824, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.7191054224967957, + "sampling/importance_sampling_ratio/max": 1.2930960655212402, + "kl": 0.01002638236968778, + "entropy": 0.0755095217609778, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.00584264844656, + "epoch": 0.00087890625, + "step": 45 + }, + { + "loss": 0.17847274243831635, + "grad_norm": 3.215383529663086, + "learning_rate": 9.102564102564102e-07, + "num_tokens": 506066.0, + "completions/mean_length": 486.75, + "completions/min_length": 167.0, + "completions/max_length": 883.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 356.8333435058594, + "completions/min_terminated_length": 167.0, + "completions/max_terminated_length": 883.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6512153148651123, + "rewards/reward_func/std": 0.13604719936847687, + "reward": 0.6512153148651123, + "reward_std": 0.13604719936847687, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020363882649689913, + "sampling/sampling_logp_difference/max": 0.33676934242248535, + "sampling/importance_sampling_ratio/min": 0.5692079663276672, + "sampling/importance_sampling_ratio/mean": 0.8801345825195312, + "sampling/importance_sampling_ratio/max": 1.305214285850525, + "kl": 0.009958896378520876, + "entropy": 0.04860728688072413, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.697979943826795, + "epoch": 0.0008984375, + "step": 46 + }, + { + "loss": 0.0022115670144557953, + "grad_norm": 3.4795665740966797, + "learning_rate": 9.076923076923076e-07, + "num_tokens": 516610.0, + "completions/mean_length": 632.875, + "completions/min_length": 186.0, + "completions/max_length": 902.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 383.0, + "completions/min_terminated_length": 186.0, + "completions/max_terminated_length": 858.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.716269850730896, + "rewards/reward_func/std": 0.10827882587909698, + "reward": 0.716269850730896, + "reward_std": 0.10827881842851639, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001914522028528154, + "sampling/sampling_logp_difference/max": 0.9404079914093018, + "sampling/importance_sampling_ratio/min": 0.317818820476532, + "sampling/importance_sampling_ratio/mean": 0.8326965570449829, + "sampling/importance_sampling_ratio/max": 1.5691876411437988, + "kl": 0.013796008803183213, + "entropy": 0.03352880047168583, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.357314236462116, + "epoch": 0.00091796875, + "step": 47 + }, + { + "loss": 0.3419533371925354, + "grad_norm": 2.572477340698242, + "learning_rate": 9.051282051282051e-07, + "num_tokens": 525356.0, + "completions/mean_length": 408.125, + "completions/min_length": 181.0, + "completions/max_length": 881.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 340.5714416503906, + "completions/min_terminated_length": 181.0, + "completions/max_terminated_length": 871.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6612143516540527, + "rewards/reward_func/std": 0.1495663970708847, + "reward": 0.6612143516540527, + "reward_std": 0.1495663821697235, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004128610715270042, + "sampling/sampling_logp_difference/max": 0.9809951782226562, + "sampling/importance_sampling_ratio/min": 0.2788654863834381, + "sampling/importance_sampling_ratio/mean": 0.7383280992507935, + "sampling/importance_sampling_ratio/max": 1.1643909215927124, + "kl": 0.010540890914853662, + "entropy": 0.06719005363993347, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.906162213534117, + "epoch": 0.0009375, + "step": 48 + }, + { + "loss": 0.1451318860054016, + "grad_norm": 4.877052307128906, + "learning_rate": 9.025641025641025e-07, + "num_tokens": 533685.0, + "completions/mean_length": 357.125, + "completions/min_length": 223.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 274.14288330078125, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 342.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5691666603088379, + "rewards/reward_func/std": 0.1307692527770996, + "reward": 0.5691666603088379, + "reward_std": 0.1307692527770996, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005150489509105682, + "sampling/sampling_logp_difference/max": 0.5296156406402588, + "sampling/importance_sampling_ratio/min": 0.39448216557502747, + "sampling/importance_sampling_ratio/mean": 0.9827626943588257, + "sampling/importance_sampling_ratio/max": 1.3634164333343506, + "kl": 0.013165580370696262, + "entropy": 0.0907683854456991, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.22686480730772, + "epoch": 0.00095703125, + "step": 49 + }, + { + "loss": -0.020071357488632202, + "grad_norm": 5.494678497314453, + "learning_rate": 9e-07, + "num_tokens": 541665.0, + "completions/mean_length": 311.375, + "completions/min_length": 163.0, + "completions/max_length": 872.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 311.375, + "completions/min_terminated_length": 163.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 7.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5366071462631226, + "rewards/reward_func/std": 0.20815543830394745, + "reward": 0.5366071462631226, + "reward_std": 0.20815545320510864, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006484447978436947, + "sampling/sampling_logp_difference/max": 1.0686261653900146, + "sampling/importance_sampling_ratio/min": 0.2569591999053955, + "sampling/importance_sampling_ratio/mean": 0.6626960635185242, + "sampling/importance_sampling_ratio/max": 1.7549552917480469, + "kl": 0.011292938914266415, + "entropy": 0.08387588011100888, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.222886353731155, + "epoch": 0.0009765625, + "step": 50 + }, + { + "loss": -0.09436193108558655, + "grad_norm": 60.68819808959961, + "learning_rate": 8.974358974358974e-07, + "num_tokens": 551012.0, + "completions/mean_length": 483.5, + "completions/min_length": 209.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 348.0, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 852.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5581597089767456, + "rewards/reward_func/std": 0.1768428534269333, + "reward": 0.5581597089767456, + "reward_std": 0.1768428385257721, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00297797960229218, + "sampling/sampling_logp_difference/max": 1.2965419292449951, + "sampling/importance_sampling_ratio/min": 0.3008386492729187, + "sampling/importance_sampling_ratio/mean": 0.6082021594047546, + "sampling/importance_sampling_ratio/max": 0.8877676129341125, + "kl": 0.016687461611581966, + "entropy": 0.05760800436837599, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.852325059473515, + "epoch": 0.00099609375, + "step": 51 + }, + { + "loss": -0.11078974604606628, + "grad_norm": 16.52714729309082, + "learning_rate": 8.948717948717949e-07, + "num_tokens": 559294.0, + "completions/mean_length": 349.5, + "completions/min_length": 192.0, + "completions/max_length": 886.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 349.5, + "completions/min_terminated_length": 192.0, + "completions/max_terminated_length": 886.0, + "tools/call_frequency": 8.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5347222089767456, + "rewards/reward_func/std": 0.2642155587673187, + "reward": 0.5347222089767456, + "reward_std": 0.2642155587673187, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005904602352529764, + "sampling/sampling_logp_difference/max": 0.9514133930206299, + "sampling/importance_sampling_ratio/min": 0.258859783411026, + "sampling/importance_sampling_ratio/mean": 0.6528693437576294, + "sampling/importance_sampling_ratio/max": 1.2452343702316284, + "kl": 0.01652154725161381, + "entropy": 0.07165615330450237, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.796393353492022, + "epoch": 0.001015625, + "step": 52 + }, + { + "loss": 0.09328754246234894, + "grad_norm": 5.860224723815918, + "learning_rate": 8.923076923076923e-07, + "num_tokens": 566617.0, + "completions/mean_length": 230.625, + "completions/min_length": 184.0, + "completions/max_length": 284.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 230.625, + "completions/min_terminated_length": 184.0, + "completions/max_terminated_length": 284.0, + "tools/call_frequency": 6.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.7687499523162842, + "rewards/reward_func/std": 0.17317645251750946, + "reward": 0.7687499523162842, + "reward_std": 0.17317645251750946, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0064454954117536545, + "sampling/sampling_logp_difference/max": 1.334283471107483, + "sampling/importance_sampling_ratio/min": 0.08370856195688248, + "sampling/importance_sampling_ratio/mean": 0.7528436779975891, + "sampling/importance_sampling_ratio/max": 1.9328172206878662, + "kl": 0.013625962659716606, + "entropy": 0.08059865748509765, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 7.25435452349484, + "epoch": 0.00103515625, + "step": 53 + }, + { + "loss": -0.13435181975364685, + "grad_norm": 1.4788209199905396, + "learning_rate": 8.897435897435897e-07, + "num_tokens": 576758.0, + "completions/mean_length": 581.875, + "completions/min_length": 203.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 386.6000061035156, + "completions/min_terminated_length": 203.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6119385957717896, + "rewards/reward_func/std": 0.21121814846992493, + "reward": 0.6119385957717896, + "reward_std": 0.21121813356876373, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002025764901190996, + "sampling/sampling_logp_difference/max": 0.37452977895736694, + "sampling/importance_sampling_ratio/min": 0.4598725140094757, + "sampling/importance_sampling_ratio/mean": 0.9685391187667847, + "sampling/importance_sampling_ratio/max": 1.466048240661621, + "kl": 0.007047314953524619, + "entropy": 0.04612128090229817, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.884415436536074, + "epoch": 0.0010546875, + "step": 54 + }, + { + "loss": 0.8524290919303894, + "grad_norm": 4.932185173034668, + "learning_rate": 8.871794871794871e-07, + "num_tokens": 585642.0, + "completions/mean_length": 425.625, + "completions/min_length": 227.0, + "completions/max_length": 871.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 425.625, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 871.0, + "tools/call_frequency": 11.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4567708373069763, + "rewards/reward_func/std": 0.25713950395584106, + "reward": 0.4567708373069763, + "reward_std": 0.25713950395584106, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004889266099780798, + "sampling/sampling_logp_difference/max": 0.9377708435058594, + "sampling/importance_sampling_ratio/min": 0.21620434522628784, + "sampling/importance_sampling_ratio/mean": 1.104311466217041, + "sampling/importance_sampling_ratio/max": 2.2183215618133545, + "kl": 0.007770946103846654, + "entropy": 0.06477197888307273, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.338715115562081, + "epoch": 0.00107421875, + "step": 55 + }, + { + "loss": -0.18761560320854187, + "grad_norm": 3.9633965492248535, + "learning_rate": 8.846153846153846e-07, + "num_tokens": 595770.0, + "completions/mean_length": 581.125, + "completions/min_length": 255.0, + "completions/max_length": 890.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 538.857177734375, + "completions/min_terminated_length": 255.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 14.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6780505776405334, + "rewards/reward_func/std": 0.2372654378414154, + "reward": 0.6780505776405334, + "reward_std": 0.2372654378414154, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024972010869532824, + "sampling/sampling_logp_difference/max": 0.47609853744506836, + "sampling/importance_sampling_ratio/min": 0.5109665989875793, + "sampling/importance_sampling_ratio/mean": 1.0867130756378174, + "sampling/importance_sampling_ratio/max": 2.04541277885437, + "kl": 0.01190740788297262, + "entropy": 0.05521496123401448, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.185026997700334, + "epoch": 0.00109375, + "step": 56 + }, + { + "loss": 0.03854670748114586, + "grad_norm": 1.775522232055664, + "learning_rate": 8.82051282051282e-07, + "num_tokens": 605969.0, + "completions/mean_length": 589.0, + "completions/min_length": 244.0, + "completions/max_length": 923.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 401.3999938964844, + "completions/min_terminated_length": 244.0, + "completions/max_terminated_length": 923.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6142247915267944, + "rewards/reward_func/std": 0.28728166222572327, + "reward": 0.6142247915267944, + "reward_std": 0.28728166222572327, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002440146403387189, + "sampling/sampling_logp_difference/max": 0.5753652453422546, + "sampling/importance_sampling_ratio/min": 0.2815491557121277, + "sampling/importance_sampling_ratio/mean": 0.7973469495773315, + "sampling/importance_sampling_ratio/max": 1.3679403066635132, + "kl": 0.010135470161912963, + "entropy": 0.05727058646152727, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.146242763847113, + "epoch": 0.00111328125, + "step": 57 + }, + { + "loss": -0.1585804671049118, + "grad_norm": 2.5361216068267822, + "learning_rate": 8.794871794871795e-07, + "num_tokens": 614916.0, + "completions/mean_length": 433.375, + "completions/min_length": 251.0, + "completions/max_length": 894.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 284.8333435058594, + "completions/min_terminated_length": 251.0, + "completions/max_terminated_length": 336.0, + "tools/call_frequency": 10.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5426042079925537, + "rewards/reward_func/std": 0.2861694395542145, + "reward": 0.5426042079925537, + "reward_std": 0.2861694395542145, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035641021095216274, + "sampling/sampling_logp_difference/max": 0.6021344661712646, + "sampling/importance_sampling_ratio/min": 0.5902630686759949, + "sampling/importance_sampling_ratio/mean": 0.9604142904281616, + "sampling/importance_sampling_ratio/max": 1.6246411800384521, + "kl": 0.013266911177197471, + "entropy": 0.06995052029378712, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.321772959083319, + "epoch": 0.0011328125, + "step": 58 + }, + { + "loss": -0.05456188693642616, + "grad_norm": 2.2564079761505127, + "learning_rate": 8.769230769230769e-07, + "num_tokens": 624397.0, + "completions/mean_length": 500.375, + "completions/min_length": 230.0, + "completions/max_length": 890.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 379.66668701171875, + "completions/min_terminated_length": 230.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 12.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4521825313568115, + "rewards/reward_func/std": 0.24139007925987244, + "reward": 0.4521825313568115, + "reward_std": 0.24139006435871124, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0036140375304967165, + "sampling/sampling_logp_difference/max": 0.4776954650878906, + "sampling/importance_sampling_ratio/min": 0.45773857831954956, + "sampling/importance_sampling_ratio/mean": 0.8759939670562744, + "sampling/importance_sampling_ratio/max": 1.4430060386657715, + "kl": 0.01054295180074405, + "entropy": 0.07078608847223222, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.158088007941842, + "epoch": 0.00115234375, + "step": 59 + }, + { + "loss": -0.21105951070785522, + "grad_norm": 1.6413570642471313, + "learning_rate": 8.743589743589743e-07, + "num_tokens": 635750.0, + "completions/mean_length": 734.0, + "completions/min_length": 288.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 647.0, + "completions/min_terminated_length": 288.0, + "completions/max_terminated_length": 893.0, + "tools/call_frequency": 17.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5532985925674438, + "rewards/reward_func/std": 0.159367173910141, + "reward": 0.5532985925674438, + "reward_std": 0.159367173910141, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001316509791649878, + "sampling/sampling_logp_difference/max": 0.9581396579742432, + "sampling/importance_sampling_ratio/min": 0.3049021363258362, + "sampling/importance_sampling_ratio/mean": 0.8415363430976868, + "sampling/importance_sampling_ratio/max": 1.4727660417556763, + "kl": 0.007134915242204443, + "entropy": 0.023028593204799108, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.702566344290972, + "epoch": 0.001171875, + "step": 60 + }, + { + "loss": 0.05024736747145653, + "grad_norm": 1.849624514579773, + "learning_rate": 8.717948717948718e-07, + "num_tokens": 646622.0, + "completions/mean_length": 672.75, + "completions/min_length": 316.0, + "completions/max_length": 924.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 556.4000244140625, + "completions/min_terminated_length": 316.0, + "completions/max_terminated_length": 924.0, + "tools/call_frequency": 16.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.546279788017273, + "rewards/reward_func/std": 0.2831685543060303, + "reward": 0.546279788017273, + "reward_std": 0.2831685245037079, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025735183153301477, + "sampling/sampling_logp_difference/max": 0.6374466419219971, + "sampling/importance_sampling_ratio/min": 0.1407802700996399, + "sampling/importance_sampling_ratio/mean": 0.7401266098022461, + "sampling/importance_sampling_ratio/max": 1.4616122245788574, + "kl": 0.005893020032090135, + "entropy": 0.05432174459565431, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.552382392808795, + "epoch": 0.00119140625, + "step": 61 + }, + { + "loss": -0.1835751235485077, + "grad_norm": 3.3736541271209717, + "learning_rate": 8.692307692307692e-07, + "num_tokens": 656795.0, + "completions/mean_length": 585.375, + "completions/min_length": 246.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 482.16668701171875, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 874.0, + "tools/call_frequency": 14.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5321022868156433, + "rewards/reward_func/std": 0.27641794085502625, + "reward": 0.5321022868156433, + "reward_std": 0.27641794085502625, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002758558839559555, + "sampling/sampling_logp_difference/max": 0.4437243938446045, + "sampling/importance_sampling_ratio/min": 0.362417995929718, + "sampling/importance_sampling_ratio/mean": 1.131781816482544, + "sampling/importance_sampling_ratio/max": 1.866432785987854, + "kl": 0.014002661599079147, + "entropy": 0.048164811974857, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.994634989649057, + "epoch": 0.0012109375, + "step": 62 + }, + { + "loss": -0.03766755759716034, + "grad_norm": 4.719907283782959, + "learning_rate": 8.666666666666667e-07, + "num_tokens": 666884.0, + "completions/mean_length": 575.75, + "completions/min_length": 189.0, + "completions/max_length": 884.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 531.7142944335938, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 14.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6739583015441895, + "rewards/reward_func/std": 0.20273123681545258, + "reward": 0.6739583015441895, + "reward_std": 0.20273125171661377, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002271033124998212, + "sampling/sampling_logp_difference/max": 0.8236135244369507, + "sampling/importance_sampling_ratio/min": 0.33007189631462097, + "sampling/importance_sampling_ratio/mean": 1.1659700870513916, + "sampling/importance_sampling_ratio/max": 2.1945693492889404, + "kl": 0.013089966640109196, + "entropy": 0.055702560988720506, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.417348457500339, + "epoch": 0.00123046875, + "step": 63 + }, + { + "loss": 0.3636564016342163, + "grad_norm": 3.988347291946411, + "learning_rate": 8.641025641025641e-07, + "num_tokens": 676356.0, + "completions/mean_length": 497.75, + "completions/min_length": 219.0, + "completions/max_length": 909.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 259.0, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 307.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.703125, + "rewards/reward_func/std": 0.2835061550140381, + "reward": 0.703125, + "reward_std": 0.2835061550140381, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002721297787502408, + "sampling/sampling_logp_difference/max": 0.4883079528808594, + "sampling/importance_sampling_ratio/min": 0.6088277101516724, + "sampling/importance_sampling_ratio/mean": 1.2220005989074707, + "sampling/importance_sampling_ratio/max": 2.911022186279297, + "kl": 0.01602328213630244, + "entropy": 0.04852636792929843, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.406257163733244, + "epoch": 0.00125, + "step": 64 + }, + { + "loss": 0.3005242943763733, + "grad_norm": 2.947273015975952, + "learning_rate": 8.615384615384616e-07, + "num_tokens": 685680.0, + "completions/mean_length": 480.0, + "completions/min_length": 238.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 430.71429443359375, + "completions/min_terminated_length": 238.0, + "completions/max_terminated_length": 880.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5356534123420715, + "rewards/reward_func/std": 0.19205154478549957, + "reward": 0.5356534123420715, + "reward_std": 0.19205154478549957, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002957387361675501, + "sampling/sampling_logp_difference/max": 0.3626279830932617, + "sampling/importance_sampling_ratio/min": 0.4092390835285187, + "sampling/importance_sampling_ratio/mean": 1.0474119186401367, + "sampling/importance_sampling_ratio/max": 1.66571044921875, + "kl": 0.01345058565493673, + "entropy": 0.07240040879696608, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.158247079700232, + "epoch": 0.00126953125, + "step": 65 + }, + { + "loss": -0.035670746117830276, + "grad_norm": 1.3001196384429932, + "learning_rate": 8.589743589743588e-07, + "num_tokens": 696267.0, + "completions/mean_length": 639.5, + "completions/min_length": 225.0, + "completions/max_length": 887.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 507.0, + "completions/min_terminated_length": 225.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 15.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.733035683631897, + "rewards/reward_func/std": 0.1327969878911972, + "reward": 0.733035683631897, + "reward_std": 0.1327969878911972, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019074230222031474, + "sampling/sampling_logp_difference/max": 0.5278514623641968, + "sampling/importance_sampling_ratio/min": 0.5020954012870789, + "sampling/importance_sampling_ratio/mean": 0.7297918796539307, + "sampling/importance_sampling_ratio/max": 1.2004237174987793, + "kl": 0.009472782199736685, + "entropy": 0.04929336876375601, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.710869509726763, + "epoch": 0.0012890625, + "step": 66 + }, + { + "loss": -0.15667179226875305, + "grad_norm": 3.5485777854919434, + "learning_rate": 8.564102564102563e-07, + "num_tokens": 706318.0, + "completions/mean_length": 572.0, + "completions/min_length": 239.0, + "completions/max_length": 885.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 527.2857666015625, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 14.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.7320684194564819, + "rewards/reward_func/std": 0.250456839799881, + "reward": 0.7320684194564819, + "reward_std": 0.2504568099975586, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002256230218335986, + "sampling/sampling_logp_difference/max": 0.2739518880844116, + "sampling/importance_sampling_ratio/min": 0.6309264302253723, + "sampling/importance_sampling_ratio/mean": 1.153450846672058, + "sampling/importance_sampling_ratio/max": 2.668430805206299, + "kl": 0.009296607895521447, + "entropy": 0.060274232644587755, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.148478474467993, + "epoch": 0.00130859375, + "step": 67 + }, + { + "loss": 1.2730926275253296, + "grad_norm": 11.827116966247559, + "learning_rate": 8.538461538461537e-07, + "num_tokens": 715523.0, + "completions/mean_length": 465.625, + "completions/min_length": 183.0, + "completions/max_length": 879.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 334.0, + "completions/min_terminated_length": 183.0, + "completions/max_terminated_length": 860.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6741319894790649, + "rewards/reward_func/std": 0.24848300218582153, + "reward": 0.6741319894790649, + "reward_std": 0.24848298728466034, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0029795509763062, + "sampling/sampling_logp_difference/max": 0.6686651706695557, + "sampling/importance_sampling_ratio/min": 0.6409409642219543, + "sampling/importance_sampling_ratio/mean": 1.2315199375152588, + "sampling/importance_sampling_ratio/max": 2.785153388977051, + "kl": 0.01075716654304415, + "entropy": 0.06116480898344889, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.867269486188889, + "epoch": 0.001328125, + "step": 68 + }, + { + "loss": -0.12663821876049042, + "grad_norm": 3.061981201171875, + "learning_rate": 8.512820512820512e-07, + "num_tokens": 724755.0, + "completions/mean_length": 469.0, + "completions/min_length": 198.0, + "completions/max_length": 888.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 411.71429443359375, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6751984357833862, + "rewards/reward_func/std": 0.3066464066505432, + "reward": 0.6751984357833862, + "reward_std": 0.3066464066505432, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0032649969216436148, + "sampling/sampling_logp_difference/max": 1.4374840259552002, + "sampling/importance_sampling_ratio/min": 0.07000722736120224, + "sampling/importance_sampling_ratio/mean": 0.9806369543075562, + "sampling/importance_sampling_ratio/max": 1.9755585193634033, + "kl": 0.019611111099948175, + "entropy": 0.05605258606374264, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.990942273288965, + "epoch": 0.00134765625, + "step": 69 + }, + { + "loss": 0.5215088129043579, + "grad_norm": 416.7526550292969, + "learning_rate": 8.487179487179486e-07, + "num_tokens": 735387.0, + "completions/mean_length": 643.125, + "completions/min_length": 214.0, + "completions/max_length": 917.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 492.6000061035156, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 898.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6476934552192688, + "rewards/reward_func/std": 0.3207628130912781, + "reward": 0.6476934552192688, + "reward_std": 0.3207627832889557, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018088469514623284, + "sampling/sampling_logp_difference/max": 0.6382637023925781, + "sampling/importance_sampling_ratio/min": 0.4919097125530243, + "sampling/importance_sampling_ratio/mean": 0.8345766067504883, + "sampling/importance_sampling_ratio/max": 1.0715612173080444, + "kl": 0.903817036858527, + "entropy": 0.03839380794670433, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.087079431861639, + "epoch": 0.0013671875, + "step": 70 + }, + { + "loss": 0.27299267053604126, + "grad_norm": 2.188009023666382, + "learning_rate": 8.461538461538461e-07, + "num_tokens": 744642.0, + "completions/mean_length": 472.125, + "completions/min_length": 202.0, + "completions/max_length": 872.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 472.125, + "completions/min_terminated_length": 202.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.671279788017273, + "rewards/reward_func/std": 0.21588072180747986, + "reward": 0.671279788017273, + "reward_std": 0.21588072180747986, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030871962662786245, + "sampling/sampling_logp_difference/max": 0.5000014901161194, + "sampling/importance_sampling_ratio/min": 0.22003255784511566, + "sampling/importance_sampling_ratio/mean": 0.9769991040229797, + "sampling/importance_sampling_ratio/max": 1.9849610328674316, + "kl": 0.008844235562719405, + "entropy": 0.06457115866942331, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.169037459418178, + "epoch": 0.00138671875, + "step": 71 + }, + { + "loss": 0.3792107105255127, + "grad_norm": 2.7811954021453857, + "learning_rate": 8.435897435897435e-07, + "num_tokens": 752724.0, + "completions/mean_length": 325.25, + "completions/min_length": 178.0, + "completions/max_length": 931.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 238.71429443359375, + "completions/min_terminated_length": 178.0, + "completions/max_terminated_length": 306.0, + "tools/call_frequency": 7.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6499255895614624, + "rewards/reward_func/std": 0.24784082174301147, + "reward": 0.6499255895614624, + "reward_std": 0.24784080684185028, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005708164069801569, + "sampling/sampling_logp_difference/max": 0.48114776611328125, + "sampling/importance_sampling_ratio/min": 0.5984382033348083, + "sampling/importance_sampling_ratio/mean": 0.9175673723220825, + "sampling/importance_sampling_ratio/max": 1.4384208917617798, + "kl": 0.016842296652612276, + "entropy": 0.0771568682976067, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.282491602003574, + "epoch": 0.00140625, + "step": 72 + }, + { + "loss": 0.00048483535647392273, + "grad_norm": 4.1132588386535645, + "learning_rate": 8.41025641025641e-07, + "num_tokens": 761517.0, + "completions/mean_length": 414.125, + "completions/min_length": 195.0, + "completions/max_length": 942.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 345.2857360839844, + "completions/min_terminated_length": 195.0, + "completions/max_terminated_length": 942.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6840277910232544, + "rewards/reward_func/std": 0.22436468303203583, + "reward": 0.6840277910232544, + "reward_std": 0.22436466813087463, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005057885777205229, + "sampling/sampling_logp_difference/max": 0.48184633255004883, + "sampling/importance_sampling_ratio/min": 0.5151718258857727, + "sampling/importance_sampling_ratio/mean": 0.9345099925994873, + "sampling/importance_sampling_ratio/max": 1.8297126293182373, + "kl": 0.014468629815382883, + "entropy": 0.08264228311600164, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.914297219365835, + "epoch": 0.00142578125, + "step": 73 + }, + { + "loss": -0.07839030772447586, + "grad_norm": 113.65328979492188, + "learning_rate": 8.384615384615384e-07, + "num_tokens": 768902.0, + "completions/mean_length": 237.875, + "completions/min_length": 198.0, + "completions/max_length": 313.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 237.875, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 313.0, + "tools/call_frequency": 6.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.7156250476837158, + "rewards/reward_func/std": 0.16476218402385712, + "reward": 0.7156250476837158, + "reward_std": 0.16476218402385712, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.008579457178711891, + "sampling/sampling_logp_difference/max": 0.9017424583435059, + "sampling/importance_sampling_ratio/min": 0.20616120100021362, + "sampling/importance_sampling_ratio/mean": 0.8562737703323364, + "sampling/importance_sampling_ratio/max": 2.1906330585479736, + "kl": 0.06715167197398841, + "entropy": 0.08828908111900091, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 7.534604227170348, + "epoch": 0.0014453125, + "step": 74 + }, + { + "loss": 0.008187921717762947, + "grad_norm": 6.224024772644043, + "learning_rate": 8.358974358974359e-07, + "num_tokens": 778927.0, + "completions/mean_length": 567.375, + "completions/min_length": 227.0, + "completions/max_length": 924.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 371.6000061035156, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 856.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6682291626930237, + "rewards/reward_func/std": 0.10036978870630264, + "reward": 0.6682291626930237, + "reward_std": 0.10036978870630264, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002103616250678897, + "sampling/sampling_logp_difference/max": 0.5457141399383545, + "sampling/importance_sampling_ratio/min": 0.23916232585906982, + "sampling/importance_sampling_ratio/mean": 0.8766606450080872, + "sampling/importance_sampling_ratio/max": 1.5751553773880005, + "kl": 0.010059219872346148, + "entropy": 0.05275670095579699, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.79867073521018, + "epoch": 0.00146484375, + "step": 75 + }, + { + "loss": 0.19994080066680908, + "grad_norm": 2.4113895893096924, + "learning_rate": 8.333333333333333e-07, + "num_tokens": 787871.0, + "completions/mean_length": 433.25, + "completions/min_length": 216.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 273.0, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 316.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6358630657196045, + "rewards/reward_func/std": 0.18506498634815216, + "reward": 0.6358630657196045, + "reward_std": 0.18506498634815216, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004888187162578106, + "sampling/sampling_logp_difference/max": 0.7505989074707031, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9192284345626831, + "sampling/importance_sampling_ratio/max": 2.6061649322509766, + "kl": 0.012994272517971694, + "entropy": 0.08687215810641646, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.9800815153867, + "epoch": 0.001484375, + "step": 76 + }, + { + "loss": -0.13543277978897095, + "grad_norm": 4.04987096786499, + "learning_rate": 8.307692307692308e-07, + "num_tokens": 796680.0, + "completions/mean_length": 415.0, + "completions/min_length": 191.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 346.0000305175781, + "completions/min_terminated_length": 191.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5418402552604675, + "rewards/reward_func/std": 0.3257177770137787, + "reward": 0.5418402552604675, + "reward_std": 0.3257177472114563, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004197363276034594, + "sampling/sampling_logp_difference/max": 1.0769401788711548, + "sampling/importance_sampling_ratio/min": 0.279384046792984, + "sampling/importance_sampling_ratio/mean": 0.6283904314041138, + "sampling/importance_sampling_ratio/max": 1.4133390188217163, + "kl": 0.01394898456055671, + "entropy": 0.07290110480971634, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.112401591613889, + "epoch": 0.00150390625, + "step": 77 + }, + { + "loss": 0.21374250948429108, + "grad_norm": 7.042120933532715, + "learning_rate": 8.282051282051282e-07, + "num_tokens": 804339.0, + "completions/mean_length": 272.25, + "completions/min_length": 207.0, + "completions/max_length": 351.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 272.25, + "completions/min_terminated_length": 207.0, + "completions/max_terminated_length": 351.0, + "tools/call_frequency": 6.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5798611044883728, + "rewards/reward_func/std": 0.1632959544658661, + "reward": 0.5798611044883728, + "reward_std": 0.1632959544658661, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00835899543017149, + "sampling/sampling_logp_difference/max": 0.5645438432693481, + "sampling/importance_sampling_ratio/min": 0.45334023237228394, + "sampling/importance_sampling_ratio/mean": 1.0002219676971436, + "sampling/importance_sampling_ratio/max": 2.1554629802703857, + "kl": 0.013974106637760997, + "entropy": 0.10557576920837164, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 7.635266935452819, + "epoch": 0.0015234375, + "step": 78 + }, + { + "loss": 0.30309754610061646, + "grad_norm": 5.299826145172119, + "learning_rate": 8.256410256410256e-07, + "num_tokens": 812916.0, + "completions/mean_length": 387.25, + "completions/min_length": 190.0, + "completions/max_length": 861.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 387.25, + "completions/min_terminated_length": 190.0, + "completions/max_terminated_length": 861.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5020833015441895, + "rewards/reward_func/std": 0.2862160801887512, + "reward": 0.5020833015441895, + "reward_std": 0.2862160801887512, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004031089134514332, + "sampling/sampling_logp_difference/max": 0.9052255153656006, + "sampling/importance_sampling_ratio/min": 0.2689155042171478, + "sampling/importance_sampling_ratio/mean": 1.038972020149231, + "sampling/importance_sampling_ratio/max": 2.182265281677246, + "kl": 0.012052874139044434, + "entropy": 0.07546961074694991, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.74857802130282, + "epoch": 0.00154296875, + "step": 79 + }, + { + "loss": 0.3059191405773163, + "grad_norm": 6.6793012619018555, + "learning_rate": 8.23076923076923e-07, + "num_tokens": 821172.0, + "completions/mean_length": 346.375, + "completions/min_length": 208.0, + "completions/max_length": 853.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 274.0, + "completions/min_terminated_length": 208.0, + "completions/max_terminated_length": 352.0, + "tools/call_frequency": 8.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5194196105003357, + "rewards/reward_func/std": 0.21844437718391418, + "reward": 0.5194196105003357, + "reward_std": 0.218444362282753, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0065603312104940414, + "sampling/sampling_logp_difference/max": 0.6814365386962891, + "sampling/importance_sampling_ratio/min": 0.30574876070022583, + "sampling/importance_sampling_ratio/mean": 1.108453631401062, + "sampling/importance_sampling_ratio/max": 2.6790430545806885, + "kl": 0.008663125277962536, + "entropy": 0.08883139200042933, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.781143864616752, + "epoch": 0.0015625, + "step": 80 + }, + { + "loss": -0.32249313592910767, + "grad_norm": 2.626694679260254, + "learning_rate": 8.205128205128205e-07, + "num_tokens": 830718.0, + "completions/mean_length": 507.625, + "completions/min_length": 239.0, + "completions/max_length": 906.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 273.3999938964844, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 306.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5825148820877075, + "rewards/reward_func/std": 0.23385532200336456, + "reward": 0.5825148820877075, + "reward_std": 0.23385529220104218, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003063466399908066, + "sampling/sampling_logp_difference/max": 0.46018171310424805, + "sampling/importance_sampling_ratio/min": 0.43080541491508484, + "sampling/importance_sampling_ratio/mean": 0.978473424911499, + "sampling/importance_sampling_ratio/max": 2.940328359603882, + "kl": 0.010730438836617395, + "entropy": 0.06515421692165546, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.616275873035192, + "epoch": 0.00158203125, + "step": 81 + }, + { + "loss": 0.057526201009750366, + "grad_norm": 2.786961793899536, + "learning_rate": 8.179487179487179e-07, + "num_tokens": 840823.0, + "completions/mean_length": 576.75, + "completions/min_length": 235.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 473.66668701171875, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 13.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5977272987365723, + "rewards/reward_func/std": 0.1760823130607605, + "reward": 0.5977272987365723, + "reward_std": 0.1760822981595993, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025689229369163513, + "sampling/sampling_logp_difference/max": 0.4526726007461548, + "sampling/importance_sampling_ratio/min": 0.4569949507713318, + "sampling/importance_sampling_ratio/mean": 1.0059870481491089, + "sampling/importance_sampling_ratio/max": 1.6006439924240112, + "kl": 0.010803742072312161, + "entropy": 0.06358908917172812, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.855118721723557, + "epoch": 0.0016015625, + "step": 82 + }, + { + "loss": 0.23709745705127716, + "grad_norm": 3.9925010204315186, + "learning_rate": 8.153846153846154e-07, + "num_tokens": 850103.0, + "completions/mean_length": 474.5, + "completions/min_length": 217.0, + "completions/max_length": 882.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 249.1999969482422, + "completions/min_terminated_length": 217.0, + "completions/max_terminated_length": 284.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6503472328186035, + "rewards/reward_func/std": 0.14649556577205658, + "reward": 0.6503472328186035, + "reward_std": 0.14649556577205658, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003622350748628378, + "sampling/sampling_logp_difference/max": 0.4952411651611328, + "sampling/importance_sampling_ratio/min": 0.5696118474006653, + "sampling/importance_sampling_ratio/mean": 0.9828578233718872, + "sampling/importance_sampling_ratio/max": 1.6415308713912964, + "kl": 0.011477014340925962, + "entropy": 0.07675067550735548, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.844152910634875, + "epoch": 0.00162109375, + "step": 83 + }, + { + "loss": -0.0907227098941803, + "grad_norm": 1.385986566543579, + "learning_rate": 8.128205128205128e-07, + "num_tokens": 860553.0, + "completions/mean_length": 620.75, + "completions/min_length": 147.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 470.6000061035156, + "completions/min_terminated_length": 147.0, + "completions/max_terminated_length": 849.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6126240491867065, + "rewards/reward_func/std": 0.15522940456867218, + "reward": 0.6126240491867065, + "reward_std": 0.155229389667511, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0031463380437344313, + "sampling/sampling_logp_difference/max": 1.4554243087768555, + "sampling/importance_sampling_ratio/min": 0.12524455785751343, + "sampling/importance_sampling_ratio/mean": 0.7162232398986816, + "sampling/importance_sampling_ratio/max": 1.4481604099273682, + "kl": 0.011507276474731043, + "entropy": 0.041387843317352235, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.04251160286367, + "epoch": 0.001640625, + "step": 84 + }, + { + "loss": 0.03867771103978157, + "grad_norm": 38.8919563293457, + "learning_rate": 8.102564102564103e-07, + "num_tokens": 869499.0, + "completions/mean_length": 432.875, + "completions/min_length": 162.0, + "completions/max_length": 906.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 282.8333435058594, + "completions/min_terminated_length": 162.0, + "completions/max_terminated_length": 326.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5296875238418579, + "rewards/reward_func/std": 0.17234690487384796, + "reward": 0.5296875238418579, + "reward_std": 0.17234688997268677, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005190685857087374, + "sampling/sampling_logp_difference/max": 1.228863000869751, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 1.018571138381958, + "sampling/importance_sampling_ratio/max": 2.034822940826416, + "kl": 0.022636780224274844, + "entropy": 0.09393353282939643, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.249907739460468, + "epoch": 0.00166015625, + "step": 85 + }, + { + "loss": 0.1577107459306717, + "grad_norm": 3.7015044689178467, + "learning_rate": 8.076923076923077e-07, + "num_tokens": 879752.0, + "completions/mean_length": 595.875, + "completions/min_length": 213.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 264.75, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 320.0, + "tools/call_frequency": 13.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.49055060744285583, + "rewards/reward_func/std": 0.28482064604759216, + "reward": 0.49055060744285583, + "reward_std": 0.28482064604759216, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025953592266887426, + "sampling/sampling_logp_difference/max": 0.6627931594848633, + "sampling/importance_sampling_ratio/min": 0.7243152260780334, + "sampling/importance_sampling_ratio/mean": 1.2104334831237793, + "sampling/importance_sampling_ratio/max": 2.074907064437866, + "kl": 0.00658827849838417, + "entropy": 0.0507073373591993, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.293472703546286, + "epoch": 0.0016796875, + "step": 86 + }, + { + "loss": 0.3262373208999634, + "grad_norm": 1.8190271854400635, + "learning_rate": 8.051282051282052e-07, + "num_tokens": 889724.0, + "completions/mean_length": 561.5, + "completions/min_length": 230.0, + "completions/max_length": 904.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 370.0, + "completions/min_terminated_length": 230.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 12.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6383432745933533, + "rewards/reward_func/std": 0.2446214258670807, + "reward": 0.6383432745933533, + "reward_std": 0.2446214109659195, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024772067554295063, + "sampling/sampling_logp_difference/max": 0.3403855562210083, + "sampling/importance_sampling_ratio/min": 0.515883207321167, + "sampling/importance_sampling_ratio/mean": 0.7886306047439575, + "sampling/importance_sampling_ratio/max": 1.3573999404907227, + "kl": 0.010086414200486615, + "entropy": 0.0532795429462567, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.666823662817478, + "epoch": 0.00169921875, + "step": 87 + }, + { + "loss": 0.40842971205711365, + "grad_norm": 2.3336265087127686, + "learning_rate": 8.025641025641025e-07, + "num_tokens": 899329.0, + "completions/mean_length": 515.5, + "completions/min_length": 224.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 270.3999938964844, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 313.0, + "tools/call_frequency": 11.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6020833253860474, + "rewards/reward_func/std": 0.24549515545368195, + "reward": 0.6020833253860474, + "reward_std": 0.24549515545368195, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0036696945317089558, + "sampling/sampling_logp_difference/max": 0.3881504535675049, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6523512005805969, + "sampling/importance_sampling_ratio/max": 1.078033685684204, + "kl": 0.01489413165836595, + "entropy": 0.07121048704721034, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.368974601849914, + "epoch": 0.00171875, + "step": 88 + }, + { + "loss": 0.6795682907104492, + "grad_norm": 4.597716808319092, + "learning_rate": 8e-07, + "num_tokens": 908100.0, + "completions/mean_length": 411.375, + "completions/min_length": 167.0, + "completions/max_length": 936.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 336.4285888671875, + "completions/min_terminated_length": 167.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5767708420753479, + "rewards/reward_func/std": 0.17966926097869873, + "reward": 0.5767708420753479, + "reward_std": 0.17966926097869873, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003954493440687656, + "sampling/sampling_logp_difference/max": 0.4081171751022339, + "sampling/importance_sampling_ratio/min": 0.5564706325531006, + "sampling/importance_sampling_ratio/mean": 1.2620627880096436, + "sampling/importance_sampling_ratio/max": 2.8474643230438232, + "kl": 0.012467298482079059, + "entropy": 0.0673908104072325, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.381834948435426, + "epoch": 0.00173828125, + "step": 89 + }, + { + "loss": 0.18657991290092468, + "grad_norm": 1.645664930343628, + "learning_rate": 7.974358974358974e-07, + "num_tokens": 917523.0, + "completions/mean_length": 493.5, + "completions/min_length": 224.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 367.8333435058594, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 879.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5223610997200012, + "rewards/reward_func/std": 0.2185569852590561, + "reward": 0.5223610997200012, + "reward_std": 0.2185569703578949, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034966622479259968, + "sampling/sampling_logp_difference/max": 0.44492673873901367, + "sampling/importance_sampling_ratio/min": 0.49509450793266296, + "sampling/importance_sampling_ratio/mean": 0.911093533039093, + "sampling/importance_sampling_ratio/max": 1.5207996368408203, + "kl": 0.010605777009914164, + "entropy": 0.07455516466870904, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.405582562088966, + "epoch": 0.0017578125, + "step": 90 + }, + { + "loss": 0.07063166797161102, + "grad_norm": 4.782991409301758, + "learning_rate": 7.948717948717948e-07, + "num_tokens": 925593.0, + "completions/mean_length": 322.125, + "completions/min_length": 210.0, + "completions/max_length": 901.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 239.4285888671875, + "completions/min_terminated_length": 210.0, + "completions/max_terminated_length": 267.0, + "tools/call_frequency": 7.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.7348958253860474, + "rewards/reward_func/std": 0.1496034562587738, + "reward": 0.7348958253860474, + "reward_std": 0.1496034562587738, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005678192712366581, + "sampling/sampling_logp_difference/max": 0.4760754108428955, + "sampling/importance_sampling_ratio/min": 0.4110691249370575, + "sampling/importance_sampling_ratio/mean": 0.959931492805481, + "sampling/importance_sampling_ratio/max": 1.5074394941329956, + "kl": 0.017596092773601413, + "entropy": 0.08636153227416798, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.69794574379921, + "epoch": 0.00177734375, + "step": 91 + }, + { + "loss": -0.01824352890253067, + "grad_norm": 2.407728433609009, + "learning_rate": 7.923076923076922e-07, + "num_tokens": 934470.0, + "completions/mean_length": 424.0, + "completions/min_length": 235.0, + "completions/max_length": 904.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 364.71429443359375, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 904.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6213541030883789, + "rewards/reward_func/std": 0.19146519899368286, + "reward": 0.6213541030883789, + "reward_std": 0.19146518409252167, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004663195461034775, + "sampling/sampling_logp_difference/max": 0.7882857322692871, + "sampling/importance_sampling_ratio/min": 0.13223344087600708, + "sampling/importance_sampling_ratio/mean": 0.8306125402450562, + "sampling/importance_sampling_ratio/max": 1.4348448514938354, + "kl": 0.011428531201090664, + "entropy": 0.08249171386705711, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.24874908849597, + "epoch": 0.001796875, + "step": 92 + }, + { + "loss": 0.013534612953662872, + "grad_norm": 5.128528118133545, + "learning_rate": 7.897435897435897e-07, + "num_tokens": 942301.0, + "completions/mean_length": 293.625, + "completions/min_length": 240.0, + "completions/max_length": 353.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 293.625, + "completions/min_terminated_length": 240.0, + "completions/max_terminated_length": 353.0, + "tools/call_frequency": 7.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47795456647872925, + "rewards/reward_func/std": 0.2416301667690277, + "reward": 0.47795456647872925, + "reward_std": 0.24163015186786652, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006663692649453878, + "sampling/sampling_logp_difference/max": 0.7301084995269775, + "sampling/importance_sampling_ratio/min": 0.4760435223579407, + "sampling/importance_sampling_ratio/mean": 0.848235011100769, + "sampling/importance_sampling_ratio/max": 1.4442698955535889, + "kl": 0.013843999186065048, + "entropy": 0.10336078982800245, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 8.066305864602327, + "epoch": 0.00181640625, + "step": 93 + }, + { + "loss": 0.052821800112724304, + "grad_norm": 3.6505239009857178, + "learning_rate": 7.871794871794871e-07, + "num_tokens": 951722.0, + "completions/mean_length": 491.625, + "completions/min_length": 205.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 430.5714416503906, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5375000238418579, + "rewards/reward_func/std": 0.22435976564884186, + "reward": 0.5375000238418579, + "reward_std": 0.22435976564884186, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003962932154536247, + "sampling/sampling_logp_difference/max": 0.8774815797805786, + "sampling/importance_sampling_ratio/min": 0.2940826416015625, + "sampling/importance_sampling_ratio/mean": 0.8591302633285522, + "sampling/importance_sampling_ratio/max": 1.7782716751098633, + "kl": 0.012335797189734876, + "entropy": 0.06003129330929369, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.185771213844419, + "epoch": 0.0018359375, + "step": 94 + }, + { + "loss": -0.05787639319896698, + "grad_norm": 4.414384841918945, + "learning_rate": 7.846153846153846e-07, + "num_tokens": 960360.0, + "completions/mean_length": 395.125, + "completions/min_length": 194.0, + "completions/max_length": 920.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 327.2857360839844, + "completions/min_terminated_length": 194.0, + "completions/max_terminated_length": 920.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6016368865966797, + "rewards/reward_func/std": 0.1689082682132721, + "reward": 0.6016368865966797, + "reward_std": 0.1689082533121109, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003714607562869787, + "sampling/sampling_logp_difference/max": 0.7719018459320068, + "sampling/importance_sampling_ratio/min": 0.5036572813987732, + "sampling/importance_sampling_ratio/mean": 0.9844136834144592, + "sampling/importance_sampling_ratio/max": 1.7979323863983154, + "kl": 0.01306924017262645, + "entropy": 0.07550884957890958, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.125007938593626, + "epoch": 0.00185546875, + "step": 95 + }, + { + "loss": 0.7171383500099182, + "grad_norm": 4.479787826538086, + "learning_rate": 7.82051282051282e-07, + "num_tokens": 969215.0, + "completions/mean_length": 422.25, + "completions/min_length": 198.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 422.25, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 934.0, + "tools/call_frequency": 9.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6015625, + "rewards/reward_func/std": 0.2524771988391876, + "reward": 0.6015625, + "reward_std": 0.25247716903686523, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003852114314213395, + "sampling/sampling_logp_difference/max": 0.4412132501602173, + "sampling/importance_sampling_ratio/min": 0.8315017819404602, + "sampling/importance_sampling_ratio/mean": 1.3867348432540894, + "sampling/importance_sampling_ratio/max": 2.3434534072875977, + "kl": 0.012307720113312826, + "entropy": 0.06837917491793633, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.073488557711244, + "epoch": 0.001875, + "step": 96 + }, + { + "loss": -0.21278181672096252, + "grad_norm": 2.2830257415771484, + "learning_rate": 7.794871794871795e-07, + "num_tokens": 978477.0, + "completions/mean_length": 472.875, + "completions/min_length": 79.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 322.0, + "completions/min_terminated_length": 79.0, + "completions/max_terminated_length": 897.0, + "tools/call_frequency": 10.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5408333539962769, + "rewards/reward_func/std": 0.3038601279258728, + "reward": 0.5408333539962769, + "reward_std": 0.3038600981235504, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025089127011597157, + "sampling/sampling_logp_difference/max": 0.4617915153503418, + "sampling/importance_sampling_ratio/min": 0.7533883452415466, + "sampling/importance_sampling_ratio/mean": 1.0878939628601074, + "sampling/importance_sampling_ratio/max": 1.5472780466079712, + "kl": 0.009183297952404246, + "entropy": 0.06731181190116331, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.095724061131477, + "epoch": 0.00189453125, + "step": 97 + }, + { + "loss": -0.036857083439826965, + "grad_norm": 5.409542560577393, + "learning_rate": 7.769230769230769e-07, + "num_tokens": 986722.0, + "completions/mean_length": 344.375, + "completions/min_length": 203.0, + "completions/max_length": 887.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 344.375, + "completions/min_terminated_length": 203.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 8.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5230158567428589, + "rewards/reward_func/std": 0.19110533595085144, + "reward": 0.5230158567428589, + "reward_std": 0.19110532104969025, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005548564717173576, + "sampling/sampling_logp_difference/max": 0.5645418167114258, + "sampling/importance_sampling_ratio/min": 0.2568794786930084, + "sampling/importance_sampling_ratio/mean": 1.0624338388442993, + "sampling/importance_sampling_ratio/max": 1.9054104089736938, + "kl": 0.016479523765156046, + "entropy": 0.0905283153988421, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.891116235405207, + "epoch": 0.0019140625, + "step": 98 + }, + { + "loss": 0.25103887915611267, + "grad_norm": 3.403409957885742, + "learning_rate": 7.743589743589744e-07, + "num_tokens": 995890.0, + "completions/mean_length": 461.75, + "completions/min_length": 212.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 315.0, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 696.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.49977678060531616, + "rewards/reward_func/std": 0.235351100564003, + "reward": 0.49977678060531616, + "reward_std": 0.2353510707616806, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030737046618014574, + "sampling/sampling_logp_difference/max": 0.42551589012145996, + "sampling/importance_sampling_ratio/min": 0.5678778886795044, + "sampling/importance_sampling_ratio/mean": 0.7660244107246399, + "sampling/importance_sampling_ratio/max": 1.2691038846969604, + "kl": 0.009272721072193235, + "entropy": 0.0660574067151174, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.753386059775949, + "epoch": 0.00193359375, + "step": 99 + }, + { + "loss": 0.7944788336753845, + "grad_norm": 4.469391822814941, + "learning_rate": 7.717948717948718e-07, + "num_tokens": 1004494.0, + "completions/mean_length": 389.75, + "completions/min_length": 193.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 317.14288330078125, + "completions/min_terminated_length": 193.0, + "completions/max_terminated_length": 852.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.703397810459137, + "rewards/reward_func/std": 0.28055569529533386, + "reward": 0.703397810459137, + "reward_std": 0.28055569529533386, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034565231762826443, + "sampling/sampling_logp_difference/max": 0.5643190145492554, + "sampling/importance_sampling_ratio/min": 0.3942466676235199, + "sampling/importance_sampling_ratio/mean": 1.1661632061004639, + "sampling/importance_sampling_ratio/max": 2.7894840240478516, + "kl": 0.013535751320887357, + "entropy": 0.06684831902384758, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.834724908694625, + "epoch": 0.001953125, + "step": 100 + }, + { + "loss": 0.057190582156181335, + "grad_norm": 2.5283045768737793, + "learning_rate": 7.692307692307693e-07, + "num_tokens": 1013277.0, + "completions/mean_length": 412.375, + "completions/min_length": 232.0, + "completions/max_length": 908.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 341.5714416503906, + "completions/min_terminated_length": 232.0, + "completions/max_terminated_length": 666.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5517897605895996, + "rewards/reward_func/std": 0.18147478997707367, + "reward": 0.5517897605895996, + "reward_std": 0.18147476017475128, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004231544211506844, + "sampling/sampling_logp_difference/max": 0.4560912847518921, + "sampling/importance_sampling_ratio/min": 0.45057791471481323, + "sampling/importance_sampling_ratio/mean": 0.9286787509918213, + "sampling/importance_sampling_ratio/max": 1.7784838676452637, + "kl": 0.011799049796536565, + "entropy": 0.07688332512043417, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.308574492111802, + "epoch": 0.00197265625, + "step": 101 + }, + { + "loss": -0.16010597348213196, + "grad_norm": 5.202837944030762, + "learning_rate": 7.666666666666667e-07, + "num_tokens": 1020954.0, + "completions/mean_length": 274.5, + "completions/min_length": 202.0, + "completions/max_length": 322.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 274.5, + "completions/min_terminated_length": 202.0, + "completions/max_terminated_length": 322.0, + "tools/call_frequency": 6.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6238541603088379, + "rewards/reward_func/std": 0.21133454144001007, + "reward": 0.6238541603088379, + "reward_std": 0.21133454144001007, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006842569913715124, + "sampling/sampling_logp_difference/max": 0.49794769287109375, + "sampling/importance_sampling_ratio/min": 0.4601878523826599, + "sampling/importance_sampling_ratio/mean": 0.8785810470581055, + "sampling/importance_sampling_ratio/max": 1.5194284915924072, + "kl": 0.01764632191043347, + "entropy": 0.1017885860055685, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 7.745818994939327, + "epoch": 0.0019921875, + "step": 102 + }, + { + "loss": -0.3362800180912018, + "grad_norm": 6.443012237548828, + "learning_rate": 7.64102564102564e-07, + "num_tokens": 1029034.0, + "completions/mean_length": 324.875, + "completions/min_length": 188.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 324.875, + "completions/min_terminated_length": 188.0, + "completions/max_terminated_length": 896.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6664930582046509, + "rewards/reward_func/std": 0.13902461528778076, + "reward": 0.6664930582046509, + "reward_std": 0.13902463018894196, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0045259022153913975, + "sampling/sampling_logp_difference/max": 0.5054498910903931, + "sampling/importance_sampling_ratio/min": 0.5449527502059937, + "sampling/importance_sampling_ratio/mean": 1.0252954959869385, + "sampling/importance_sampling_ratio/max": 2.183760404586792, + "kl": 0.0129995311726816, + "entropy": 0.08502721652621403, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.122416742146015, + "epoch": 0.00201171875, + "step": 103 + }, + { + "loss": 0.02754916250705719, + "grad_norm": 5.346973419189453, + "learning_rate": 7.615384615384615e-07, + "num_tokens": 1036744.0, + "completions/mean_length": 279.375, + "completions/min_length": 216.0, + "completions/max_length": 355.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 279.375, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 355.0, + "tools/call_frequency": 6.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5208333730697632, + "rewards/reward_func/std": 0.15402987599372864, + "reward": 0.5208333730697632, + "reward_std": 0.15402986109256744, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.007777934893965721, + "sampling/sampling_logp_difference/max": 1.0369467735290527, + "sampling/importance_sampling_ratio/min": 0.3366830050945282, + "sampling/importance_sampling_ratio/mean": 0.6198764443397522, + "sampling/importance_sampling_ratio/max": 1.1137953996658325, + "kl": 0.016595231485553086, + "entropy": 0.10079135000705719, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 8.138191860169172, + "epoch": 0.00203125, + "step": 104 + }, + { + "loss": 0.04734654724597931, + "grad_norm": 4.402525901794434, + "learning_rate": 7.589743589743589e-07, + "num_tokens": 1044168.0, + "completions/mean_length": 243.25, + "completions/min_length": 213.0, + "completions/max_length": 288.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 243.25, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 288.0, + "tools/call_frequency": 5.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.671875, + "rewards/reward_func/std": 0.1423744112253189, + "reward": 0.671875, + "reward_std": 0.14237439632415771, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.010006429627537727, + "sampling/sampling_logp_difference/max": 0.8482804298400879, + "sampling/importance_sampling_ratio/min": 0.41829702258110046, + "sampling/importance_sampling_ratio/mean": 0.6914764642715454, + "sampling/importance_sampling_ratio/max": 1.3068418502807617, + "kl": 0.015844520297832787, + "entropy": 0.1027985205873847, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.808194316923618, + "epoch": 0.00205078125, + "step": 105 + }, + { + "loss": 0.23303216695785522, + "grad_norm": 3.2454283237457275, + "learning_rate": 7.564102564102564e-07, + "num_tokens": 1052962.0, + "completions/mean_length": 413.625, + "completions/min_length": 229.0, + "completions/max_length": 879.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 413.625, + "completions/min_terminated_length": 229.0, + "completions/max_terminated_length": 879.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5833333134651184, + "rewards/reward_func/std": 0.14692619442939758, + "reward": 0.5833333134651184, + "reward_std": 0.1469261646270752, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003668921533972025, + "sampling/sampling_logp_difference/max": 0.5063203573226929, + "sampling/importance_sampling_ratio/min": 0.5988433957099915, + "sampling/importance_sampling_ratio/mean": 1.0708799362182617, + "sampling/importance_sampling_ratio/max": 1.7189308404922485, + "kl": 0.010812907421495765, + "entropy": 0.07479791343212128, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.99059883877635, + "epoch": 0.0020703125, + "step": 106 + }, + { + "loss": 0.08047209680080414, + "grad_norm": 3.486604928970337, + "learning_rate": 7.538461538461538e-07, + "num_tokens": 1061647.0, + "completions/mean_length": 400.625, + "completions/min_length": 197.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 244.83334350585938, + "completions/min_terminated_length": 197.0, + "completions/max_terminated_length": 296.0, + "tools/call_frequency": 10.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5388392806053162, + "rewards/reward_func/std": 0.24130317568778992, + "reward": 0.5388392806053162, + "reward_std": 0.24130316078662872, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004338262137025595, + "sampling/sampling_logp_difference/max": 0.44372355937957764, + "sampling/importance_sampling_ratio/min": 0.33691373467445374, + "sampling/importance_sampling_ratio/mean": 0.7600634098052979, + "sampling/importance_sampling_ratio/max": 1.9062042236328125, + "kl": 0.011786237300839275, + "entropy": 0.06849817262263969, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.883897544816136, + "epoch": 0.00208984375, + "step": 107 + }, + { + "loss": 0.059548765420913696, + "grad_norm": 4.282438278198242, + "learning_rate": 7.512820512820513e-07, + "num_tokens": 1070262.0, + "completions/mean_length": 391.875, + "completions/min_length": 220.0, + "completions/max_length": 888.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 391.875, + "completions/min_terminated_length": 220.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6256200075149536, + "rewards/reward_func/std": 0.2583758533000946, + "reward": 0.6256200075149536, + "reward_std": 0.2583758533000946, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004532008897513151, + "sampling/sampling_logp_difference/max": 1.3009966611862183, + "sampling/importance_sampling_ratio/min": 0.08350058645009995, + "sampling/importance_sampling_ratio/mean": 0.882885217666626, + "sampling/importance_sampling_ratio/max": 1.4893348217010498, + "kl": 0.013987419923068956, + "entropy": 0.07234736543614417, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.636090299114585, + "epoch": 0.002109375, + "step": 108 + }, + { + "loss": 0.2815035879611969, + "grad_norm": 23.74864387512207, + "learning_rate": 7.487179487179486e-07, + "num_tokens": 1079181.0, + "completions/mean_length": 429.5, + "completions/min_length": 220.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 268.8333435058594, + "completions/min_terminated_length": 220.0, + "completions/max_terminated_length": 291.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.460833340883255, + "rewards/reward_func/std": 0.30305883288383484, + "reward": 0.460833340883255, + "reward_std": 0.30305880308151245, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004240637645125389, + "sampling/sampling_logp_difference/max": 0.7246110439300537, + "sampling/importance_sampling_ratio/min": 0.5076007843017578, + "sampling/importance_sampling_ratio/mean": 0.9406247138977051, + "sampling/importance_sampling_ratio/max": 2.3668487071990967, + "kl": 0.03918976685963571, + "entropy": 0.08903116825968027, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.988295014947653, + "epoch": 0.00212890625, + "step": 109 + }, + { + "loss": 0.024441178888082504, + "grad_norm": 4.709048748016357, + "learning_rate": 7.461538461538461e-07, + "num_tokens": 1088145.0, + "completions/mean_length": 434.5, + "completions/min_length": 222.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 278.5, + "completions/min_terminated_length": 222.0, + "completions/max_terminated_length": 317.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4910416603088379, + "rewards/reward_func/std": 0.1269848346710205, + "reward": 0.4910416603088379, + "reward_std": 0.1269848346710205, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00440294248983264, + "sampling/sampling_logp_difference/max": 0.4926304817199707, + "sampling/importance_sampling_ratio/min": 0.4826917350292206, + "sampling/importance_sampling_ratio/mean": 1.178555965423584, + "sampling/importance_sampling_ratio/max": 2.509007215499878, + "kl": 0.018098352069500834, + "entropy": 0.08046847372315824, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.93582141213119, + "epoch": 0.0021484375, + "step": 110 + }, + { + "loss": -0.4657578468322754, + "grad_norm": 5.248799800872803, + "learning_rate": 7.435897435897435e-07, + "num_tokens": 1096271.0, + "completions/mean_length": 331.25, + "completions/min_length": 98.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 250.57144165039062, + "completions/min_terminated_length": 98.0, + "completions/max_terminated_length": 342.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5434564352035522, + "rewards/reward_func/std": 0.3118937611579895, + "reward": 0.5434564352035522, + "reward_std": 0.3118937313556671, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005716612562537193, + "sampling/sampling_logp_difference/max": 0.49338626861572266, + "sampling/importance_sampling_ratio/min": 0.5145726203918457, + "sampling/importance_sampling_ratio/mean": 1.0240150690078735, + "sampling/importance_sampling_ratio/max": 1.8040510416030884, + "kl": 0.012770107401593123, + "entropy": 0.0834533182787709, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.886128542944789, + "epoch": 0.00216796875, + "step": 111 + }, + { + "loss": -0.09941476583480835, + "grad_norm": 5.677592754364014, + "learning_rate": 7.41025641025641e-07, + "num_tokens": 1104372.0, + "completions/mean_length": 326.25, + "completions/min_length": 205.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 244.71429443359375, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 327.0, + "tools/call_frequency": 7.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6315972208976746, + "rewards/reward_func/std": 0.22497273981571198, + "reward": 0.6315972208976746, + "reward_std": 0.22497273981571198, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005961448419839144, + "sampling/sampling_logp_difference/max": 1.2170718908309937, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6739175915718079, + "sampling/importance_sampling_ratio/max": 1.797560691833496, + "kl": 0.013760226662270725, + "entropy": 0.08792887371964753, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.816299324855208, + "epoch": 0.0021875, + "step": 112 + }, + { + "loss": -0.1611029952764511, + "grad_norm": 2.1033620834350586, + "learning_rate": 7.384615384615384e-07, + "num_tokens": 1113136.0, + "completions/mean_length": 410.125, + "completions/min_length": 220.0, + "completions/max_length": 878.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 349.4285888671875, + "completions/min_terminated_length": 220.0, + "completions/max_terminated_length": 878.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5899147987365723, + "rewards/reward_func/std": 0.2545212209224701, + "reward": 0.5899147987365723, + "reward_std": 0.2545212209224701, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004053469281643629, + "sampling/sampling_logp_difference/max": 0.6650888919830322, + "sampling/importance_sampling_ratio/min": 0.2904740571975708, + "sampling/importance_sampling_ratio/mean": 0.8569414615631104, + "sampling/importance_sampling_ratio/max": 1.3320674896240234, + "kl": 0.008077691396465525, + "entropy": 0.07181810960173607, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.750642273575068, + "epoch": 0.00220703125, + "step": 113 + }, + { + "loss": 0.04018804803490639, + "grad_norm": 1.7398210763931274, + "learning_rate": 7.358974358974359e-07, + "num_tokens": 1123556.0, + "completions/mean_length": 616.125, + "completions/min_length": 277.0, + "completions/max_length": 955.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 430.3999938964844, + "completions/min_terminated_length": 277.0, + "completions/max_terminated_length": 878.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4385416507720947, + "rewards/reward_func/std": 0.12710042297840118, + "reward": 0.4385416507720947, + "reward_std": 0.12710042297840118, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003395616076886654, + "sampling/sampling_logp_difference/max": 0.43999409675598145, + "sampling/importance_sampling_ratio/min": 0.24526093900203705, + "sampling/importance_sampling_ratio/mean": 0.6846455335617065, + "sampling/importance_sampling_ratio/max": 1.0143914222717285, + "kl": 0.008938136423239484, + "entropy": 0.07033037871588022, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.156013077124953, + "epoch": 0.0022265625, + "step": 114 + }, + { + "loss": 0.13914915919303894, + "grad_norm": 2.9167020320892334, + "learning_rate": 7.333333333333332e-07, + "num_tokens": 1131925.0, + "completions/mean_length": 361.875, + "completions/min_length": 215.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 287.8571472167969, + "completions/min_terminated_length": 215.0, + "completions/max_terminated_length": 403.0, + "tools/call_frequency": 8.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6088541746139526, + "rewards/reward_func/std": 0.27681970596313477, + "reward": 0.6088541746139526, + "reward_std": 0.27681970596313477, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004696145188063383, + "sampling/sampling_logp_difference/max": 0.4390699565410614, + "sampling/importance_sampling_ratio/min": 0.554343044757843, + "sampling/importance_sampling_ratio/mean": 1.0064725875854492, + "sampling/importance_sampling_ratio/max": 1.5421404838562012, + "kl": 0.010933810437563807, + "entropy": 0.07991569873411208, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.362923027947545, + "epoch": 0.00224609375, + "step": 115 + }, + { + "loss": 0.022982129827141762, + "grad_norm": 2.6425719261169434, + "learning_rate": 7.307692307692307e-07, + "num_tokens": 1140939.0, + "completions/mean_length": 441.125, + "completions/min_length": 253.0, + "completions/max_length": 888.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 380.14288330078125, + "completions/min_terminated_length": 253.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 10.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4902033805847168, + "rewards/reward_func/std": 0.22059309482574463, + "reward": 0.4902033805847168, + "reward_std": 0.22059309482574463, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003999285399913788, + "sampling/sampling_logp_difference/max": 1.0842822790145874, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.58584064245224, + "sampling/importance_sampling_ratio/max": 1.2791460752487183, + "kl": 0.010381651314673945, + "entropy": 0.07733333413489163, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.06179778277874, + "epoch": 0.002265625, + "step": 116 + }, + { + "loss": 0.0656152069568634, + "grad_norm": 14.95664119720459, + "learning_rate": 7.282051282051281e-07, + "num_tokens": 1150526.0, + "completions/mean_length": 513.75, + "completions/min_length": 294.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 316.3999938964844, + "completions/min_terminated_length": 294.0, + "completions/max_terminated_length": 343.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5729166865348816, + "rewards/reward_func/std": 0.3034687042236328, + "reward": 0.5729166865348816, + "reward_std": 0.3034687340259552, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034258561208844185, + "sampling/sampling_logp_difference/max": 0.6278234720230103, + "sampling/importance_sampling_ratio/min": 0.5120461583137512, + "sampling/importance_sampling_ratio/mean": 0.8304383754730225, + "sampling/importance_sampling_ratio/max": 1.2600904703140259, + "kl": 0.0143667227239348, + "entropy": 0.07604050438385457, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.227332254871726, + "epoch": 0.00228515625, + "step": 117 + }, + { + "loss": 0.04637216031551361, + "grad_norm": 3.5778422355651855, + "learning_rate": 7.256410256410256e-07, + "num_tokens": 1160067.0, + "completions/mean_length": 506.625, + "completions/min_length": 242.0, + "completions/max_length": 910.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 374.0, + "completions/min_terminated_length": 242.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5782986283302307, + "rewards/reward_func/std": 0.17787067592144012, + "reward": 0.5782986283302307, + "reward_std": 0.17787067592144012, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003001322504132986, + "sampling/sampling_logp_difference/max": 0.730076789855957, + "sampling/importance_sampling_ratio/min": 0.44296330213546753, + "sampling/importance_sampling_ratio/mean": 0.9703428745269775, + "sampling/importance_sampling_ratio/max": 1.7808029651641846, + "kl": 0.009051153145264834, + "entropy": 0.06876774644479156, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.024364059790969, + "epoch": 0.0023046875, + "step": 118 + }, + { + "loss": -0.2600385546684265, + "grad_norm": 4.365750789642334, + "learning_rate": 7.23076923076923e-07, + "num_tokens": 1170234.0, + "completions/mean_length": 585.125, + "completions/min_length": 223.0, + "completions/max_length": 923.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 499.66668701171875, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 923.0, + "tools/call_frequency": 14.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5439867377281189, + "rewards/reward_func/std": 0.16924212872982025, + "reward": 0.5439867377281189, + "reward_std": 0.16924212872982025, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003094877814874053, + "sampling/sampling_logp_difference/max": 0.737809419631958, + "sampling/importance_sampling_ratio/min": 0.3791033923625946, + "sampling/importance_sampling_ratio/mean": 1.1130623817443848, + "sampling/importance_sampling_ratio/max": 1.9192607402801514, + "kl": 0.016095810802653432, + "entropy": 0.062278405646793544, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.695310533046722, + "epoch": 0.00232421875, + "step": 119 + }, + { + "loss": 0.2350778728723526, + "grad_norm": 4.665589332580566, + "learning_rate": 7.205128205128205e-07, + "num_tokens": 1179834.0, + "completions/mean_length": 515.375, + "completions/min_length": 227.0, + "completions/max_length": 907.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 388.16668701171875, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6349697709083557, + "rewards/reward_func/std": 0.2329857498407364, + "reward": 0.6349697709083557, + "reward_std": 0.23298576474189758, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0036732456646859646, + "sampling/sampling_logp_difference/max": 0.4825303554534912, + "sampling/importance_sampling_ratio/min": 0.6615427732467651, + "sampling/importance_sampling_ratio/mean": 1.239565134048462, + "sampling/importance_sampling_ratio/max": 2.642176389694214, + "kl": 0.014938207867089659, + "entropy": 0.060711000580340624, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.488409642130136, + "epoch": 0.00234375, + "step": 120 + }, + { + "loss": -0.212884783744812, + "grad_norm": 1.0752605199813843, + "learning_rate": 7.179487179487179e-07, + "num_tokens": 1190938.0, + "completions/mean_length": 702.0, + "completions/min_length": 259.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 590.0, + "completions/min_terminated_length": 259.0, + "completions/max_terminated_length": 880.0, + "tools/call_frequency": 18.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5999603271484375, + "rewards/reward_func/std": 0.3016301393508911, + "reward": 0.5999603271484375, + "reward_std": 0.3016301095485687, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0022001308389008045, + "sampling/sampling_logp_difference/max": 0.3981378674507141, + "sampling/importance_sampling_ratio/min": 0.3063691258430481, + "sampling/importance_sampling_ratio/mean": 0.8632944822311401, + "sampling/importance_sampling_ratio/max": 1.3224005699157715, + "kl": 0.012537860369775444, + "entropy": 0.04799047764390707, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.569462010636926, + "epoch": 0.00236328125, + "step": 121 + }, + { + "loss": 0.1214713305234909, + "grad_norm": 1.575884461402893, + "learning_rate": 7.153846153846154e-07, + "num_tokens": 1202951.0, + "completions/mean_length": 816.25, + "completions/min_length": 264.0, + "completions/max_length": 936.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 572.5, + "completions/min_terminated_length": 264.0, + "completions/max_terminated_length": 881.0, + "tools/call_frequency": 18.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.601784348487854, + "rewards/reward_func/std": 0.24771520495414734, + "reward": 0.601784348487854, + "reward_std": 0.24771520495414734, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0012568383244797587, + "sampling/sampling_logp_difference/max": 0.5278418064117432, + "sampling/importance_sampling_ratio/min": 0.43986716866493225, + "sampling/importance_sampling_ratio/mean": 0.9192232489585876, + "sampling/importance_sampling_ratio/max": 1.5081967115402222, + "kl": 0.007006631727563217, + "entropy": 0.02481439191615209, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.416979398578405, + "epoch": 0.0023828125, + "step": 122 + }, + { + "loss": -0.17005760967731476, + "grad_norm": 1.5449719429016113, + "learning_rate": 7.128205128205128e-07, + "num_tokens": 1214177.0, + "completions/mean_length": 719.125, + "completions/min_length": 236.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 477.0, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 865.0, + "tools/call_frequency": 17.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6095486283302307, + "rewards/reward_func/std": 0.285736620426178, + "reward": 0.6095486283302307, + "reward_std": 0.285736620426178, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021787239238619804, + "sampling/sampling_logp_difference/max": 0.47433996200561523, + "sampling/importance_sampling_ratio/min": 0.47243189811706543, + "sampling/importance_sampling_ratio/mean": 1.0789601802825928, + "sampling/importance_sampling_ratio/max": 1.7422783374786377, + "kl": 0.00937717875058297, + "entropy": 0.035182658524718136, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.78382838703692, + "epoch": 0.00240234375, + "step": 123 + }, + { + "loss": -0.12292590737342834, + "grad_norm": 3.687173366546631, + "learning_rate": 7.102564102564103e-07, + "num_tokens": 1224926.0, + "completions/mean_length": 658.375, + "completions/min_length": 241.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 433.75, + "completions/min_terminated_length": 241.0, + "completions/max_terminated_length": 876.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.628993034362793, + "rewards/reward_func/std": 0.23762406408786774, + "reward": 0.628993034362793, + "reward_std": 0.23762404918670654, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024015705566853285, + "sampling/sampling_logp_difference/max": 0.8136589527130127, + "sampling/importance_sampling_ratio/min": 0.4155843257904053, + "sampling/importance_sampling_ratio/mean": 1.1059954166412354, + "sampling/importance_sampling_ratio/max": 2.8862545490264893, + "kl": 0.008897862979210913, + "entropy": 0.04466253420105204, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 15.565726786851883, + "epoch": 0.002421875, + "step": 124 + }, + { + "loss": 0.27603381872177124, + "grad_norm": 2.950641632080078, + "learning_rate": 7.076923076923077e-07, + "num_tokens": 1235806.0, + "completions/mean_length": 674.5, + "completions/min_length": 239.0, + "completions/max_length": 976.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 543.7999877929688, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 976.0, + "tools/call_frequency": 14.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4642992317676544, + "rewards/reward_func/std": 0.14427730441093445, + "reward": 0.4642992317676544, + "reward_std": 0.14427730441093445, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002082474995404482, + "sampling/sampling_logp_difference/max": 0.34878015518188477, + "sampling/importance_sampling_ratio/min": 0.6374633312225342, + "sampling/importance_sampling_ratio/mean": 1.0793042182922363, + "sampling/importance_sampling_ratio/max": 1.7116844654083252, + "kl": 0.011601899881497957, + "entropy": 0.043801160994917154, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 17.249050250276923, + "epoch": 0.00244140625, + "step": 125 + }, + { + "loss": -0.1113254725933075, + "grad_norm": 2.2773635387420654, + "learning_rate": 7.051282051282052e-07, + "num_tokens": 1246530.0, + "completions/mean_length": 654.625, + "completions/min_length": 227.0, + "completions/max_length": 903.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 622.1428833007812, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 903.0, + "tools/call_frequency": 15.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6389881372451782, + "rewards/reward_func/std": 0.2807202637195587, + "reward": 0.6389881372451782, + "reward_std": 0.28072023391723633, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020052988547831774, + "sampling/sampling_logp_difference/max": 0.4909176826477051, + "sampling/importance_sampling_ratio/min": 0.4408179521560669, + "sampling/importance_sampling_ratio/mean": 1.0227587223052979, + "sampling/importance_sampling_ratio/max": 1.6894110441207886, + "kl": 0.013891706970753148, + "entropy": 0.036761358118383214, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.073368284851313, + "epoch": 0.0024609375, + "step": 126 + }, + { + "loss": -0.0015081241726875305, + "grad_norm": 3.2780792713165283, + "learning_rate": 7.025641025641025e-07, + "num_tokens": 1258341.0, + "completions/mean_length": 790.875, + "completions/min_length": 258.0, + "completions/max_length": 889.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 673.0, + "completions/min_terminated_length": 258.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 19.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.7301339507102966, + "rewards/reward_func/std": 0.1350851207971573, + "reward": 0.7301339507102966, + "reward_std": 0.13508513569831848, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001390608143992722, + "sampling/sampling_logp_difference/max": 0.4617612361907959, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 1.2181260585784912, + "sampling/importance_sampling_ratio/max": 2.592003107070923, + "kl": 0.007393870531814173, + "entropy": 0.021450561995152384, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.06309207342565, + "epoch": 0.00248046875, + "step": 127 + }, + { + "loss": 0.1538911908864975, + "grad_norm": 1.5431100130081177, + "learning_rate": 7e-07, + "num_tokens": 1270167.0, + "completions/mean_length": 792.5, + "completions/min_length": 193.0, + "completions/max_length": 909.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 658.3333740234375, + "completions/min_terminated_length": 193.0, + "completions/max_terminated_length": 909.0, + "tools/call_frequency": 18.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6985998153686523, + "rewards/reward_func/std": 0.20408636331558228, + "reward": 0.6985998153686523, + "reward_std": 0.20408634841442108, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001003478653728962, + "sampling/sampling_logp_difference/max": 0.22853851318359375, + "sampling/importance_sampling_ratio/min": 0.6358345150947571, + "sampling/importance_sampling_ratio/mean": 1.0301367044448853, + "sampling/importance_sampling_ratio/max": 1.7454187870025635, + "kl": 0.00789379610796459, + "entropy": 0.02081299087149091, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.26056282222271, + "epoch": 0.0025, + "step": 128 + }, + { + "loss": -0.20812654495239258, + "grad_norm": 1.2758002281188965, + "learning_rate": 6.974358974358974e-07, + "num_tokens": 1281063.0, + "completions/mean_length": 676.375, + "completions/min_length": 241.0, + "completions/max_length": 902.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 546.2000122070312, + "completions/min_terminated_length": 241.0, + "completions/max_terminated_length": 902.0, + "tools/call_frequency": 15.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6836061477661133, + "rewards/reward_func/std": 0.1218762993812561, + "reward": 0.6836061477661133, + "reward_std": 0.1218762993812561, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018409335752949119, + "sampling/sampling_logp_difference/max": 0.9900542497634888, + "sampling/importance_sampling_ratio/min": 0.568325400352478, + "sampling/importance_sampling_ratio/mean": 0.995194673538208, + "sampling/importance_sampling_ratio/max": 1.4416425228118896, + "kl": 0.006820491136750206, + "entropy": 0.044757036928785965, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.330166218802333, + "epoch": 0.00251953125, + "step": 129 + }, + { + "loss": -0.20104463398456573, + "grad_norm": 1.7743576765060425, + "learning_rate": 6.948717948717948e-07, + "num_tokens": 1292253.0, + "completions/mean_length": 712.75, + "completions/min_length": 239.0, + "completions/max_length": 914.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 554.25, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 17.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6267361640930176, + "rewards/reward_func/std": 0.19507233798503876, + "reward": 0.6267361640930176, + "reward_std": 0.19507233798503876, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0015519054140895605, + "sampling/sampling_logp_difference/max": 0.7147161960601807, + "sampling/importance_sampling_ratio/min": 0.5341469049453735, + "sampling/importance_sampling_ratio/mean": 1.1017919778823853, + "sampling/importance_sampling_ratio/max": 2.524951457977295, + "kl": 0.012166208762209862, + "entropy": 0.03314427169971168, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.990468643605709, + "epoch": 0.0025390625, + "step": 130 + }, + { + "loss": -0.04797378554940224, + "grad_norm": 2.749682903289795, + "learning_rate": 6.923076923076922e-07, + "num_tokens": 1304721.0, + "completions/mean_length": 874.125, + "completions/min_length": 834.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 857.6666870117188, + "completions/min_terminated_length": 834.0, + "completions/max_terminated_length": 895.0, + "tools/call_frequency": 21.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5390625, + "rewards/reward_func/std": 0.28983214497566223, + "reward": 0.5390625, + "reward_std": 0.28983214497566223, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011467327130958438, + "sampling/sampling_logp_difference/max": 0.4883997440338135, + "sampling/importance_sampling_ratio/min": 0.36340129375457764, + "sampling/importance_sampling_ratio/mean": 0.7015590071678162, + "sampling/importance_sampling_ratio/max": 1.072097659111023, + "kl": 0.007361818687058985, + "entropy": 0.010802999400766566, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.902000060305, + "epoch": 0.00255859375, + "step": 131 + }, + { + "loss": 0.09429103136062622, + "grad_norm": 1.896155595779419, + "learning_rate": 6.897435897435897e-07, + "num_tokens": 1314703.0, + "completions/mean_length": 562.5, + "completions/min_length": 209.0, + "completions/max_length": 940.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 462.5, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 940.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6236110925674438, + "rewards/reward_func/std": 0.2858586609363556, + "reward": 0.6236110925674438, + "reward_std": 0.2858586311340332, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002391051035374403, + "sampling/sampling_logp_difference/max": 0.4654519557952881, + "sampling/importance_sampling_ratio/min": 0.7202103734016418, + "sampling/importance_sampling_ratio/mean": 0.8500819802284241, + "sampling/importance_sampling_ratio/max": 0.9952996373176575, + "kl": 0.010915465507423505, + "entropy": 0.05931007061735727, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.541412929072976, + "epoch": 0.002578125, + "step": 132 + }, + { + "loss": 0.1038965955376625, + "grad_norm": 2.86712646484375, + "learning_rate": 6.871794871794871e-07, + "num_tokens": 1325922.0, + "completions/mean_length": 718.0, + "completions/min_length": 218.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 453.3333435058594, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 897.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6784722805023193, + "rewards/reward_func/std": 0.290201336145401, + "reward": 0.6784722805023193, + "reward_std": 0.290201336145401, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013128025457262993, + "sampling/sampling_logp_difference/max": 0.5928339958190918, + "sampling/importance_sampling_ratio/min": 0.791932225227356, + "sampling/importance_sampling_ratio/mean": 1.1047841310501099, + "sampling/importance_sampling_ratio/max": 1.5196712017059326, + "kl": 0.010374652949394658, + "entropy": 0.029705540917348117, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.223529336974025, + "epoch": 0.00259765625, + "step": 133 + }, + { + "loss": 0.08396722376346588, + "grad_norm": 2.442532777786255, + "learning_rate": 6.846153846153846e-07, + "num_tokens": 1336559.0, + "completions/mean_length": 645.0, + "completions/min_length": 212.0, + "completions/max_length": 948.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 267.0, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 324.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6925594806671143, + "rewards/reward_func/std": 0.16971668601036072, + "reward": 0.6925594806671143, + "reward_std": 0.16971668601036072, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025252806954085827, + "sampling/sampling_logp_difference/max": 0.39813435077667236, + "sampling/importance_sampling_ratio/min": 0.170582115650177, + "sampling/importance_sampling_ratio/mean": 0.7900192141532898, + "sampling/importance_sampling_ratio/max": 1.329661250114441, + "kl": 0.009932571032550186, + "entropy": 0.04072210076265037, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.834926316514611, + "epoch": 0.0026171875, + "step": 134 + }, + { + "loss": 0.03690458834171295, + "grad_norm": 2.528012752532959, + "learning_rate": 6.82051282051282e-07, + "num_tokens": 1347197.0, + "completions/mean_length": 645.0, + "completions/min_length": 238.0, + "completions/max_length": 904.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 398.25, + "completions/min_terminated_length": 238.0, + "completions/max_terminated_length": 831.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5713731050491333, + "rewards/reward_func/std": 0.19196496903896332, + "reward": 0.5713731050491333, + "reward_std": 0.19196495413780212, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019870216492563486, + "sampling/sampling_logp_difference/max": 0.6007534265518188, + "sampling/importance_sampling_ratio/min": 0.3059462606906891, + "sampling/importance_sampling_ratio/mean": 1.0975478887557983, + "sampling/importance_sampling_ratio/max": 2.477039337158203, + "kl": 0.007351052539888769, + "entropy": 0.044850681093521416, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.126545671373606, + "epoch": 0.00263671875, + "step": 135 + }, + { + "loss": 0.21726936101913452, + "grad_norm": 4.2412190437316895, + "learning_rate": 6.794871794871795e-07, + "num_tokens": 1356092.0, + "completions/mean_length": 427.625, + "completions/min_length": 222.0, + "completions/max_length": 899.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 274.3333435058594, + "completions/min_terminated_length": 222.0, + "completions/max_terminated_length": 321.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6865909099578857, + "rewards/reward_func/std": 0.20461229979991913, + "reward": 0.6865909099578857, + "reward_std": 0.20461229979991913, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0039690579287707806, + "sampling/sampling_logp_difference/max": 0.5126842260360718, + "sampling/importance_sampling_ratio/min": 0.41742751002311707, + "sampling/importance_sampling_ratio/mean": 0.9628218412399292, + "sampling/importance_sampling_ratio/max": 2.1050844192504883, + "kl": 0.010727933695307001, + "entropy": 0.07418314676033333, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.068690648302436, + "epoch": 0.00265625, + "step": 136 + }, + { + "loss": 0.44740206003189087, + "grad_norm": 3.361337184906006, + "learning_rate": 6.769230769230769e-07, + "num_tokens": 1367462.0, + "completions/mean_length": 735.875, + "completions/min_length": 256.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 260.5, + "completions/min_terminated_length": 256.0, + "completions/max_terminated_length": 265.0, + "tools/call_frequency": 17.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5557291507720947, + "rewards/reward_func/std": 0.33560192584991455, + "reward": 0.5557291507720947, + "reward_std": 0.33560192584991455, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011866922723129392, + "sampling/sampling_logp_difference/max": 0.4676344394683838, + "sampling/importance_sampling_ratio/min": 0.5687475204467773, + "sampling/importance_sampling_ratio/mean": 1.1488490104675293, + "sampling/importance_sampling_ratio/max": 2.3437399864196777, + "kl": 0.007502066822780762, + "entropy": 0.03031673567602411, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.77722679451108, + "epoch": 0.00267578125, + "step": 137 + }, + { + "loss": 0.418165922164917, + "grad_norm": 3.736605644226074, + "learning_rate": 6.743589743589744e-07, + "num_tokens": 1377360.0, + "completions/mean_length": 552.125, + "completions/min_length": 233.0, + "completions/max_length": 922.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 349.20001220703125, + "completions/min_terminated_length": 233.0, + "completions/max_terminated_length": 806.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5405505895614624, + "rewards/reward_func/std": 0.281326562166214, + "reward": 0.5405505895614624, + "reward_std": 0.281326562166214, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034071423579007387, + "sampling/sampling_logp_difference/max": 0.9259915351867676, + "sampling/importance_sampling_ratio/min": 0.29359719157218933, + "sampling/importance_sampling_ratio/mean": 1.0875670909881592, + "sampling/importance_sampling_ratio/max": 1.9160518646240234, + "kl": 0.013977142429212108, + "entropy": 0.07162515202071518, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.074734007939696, + "epoch": 0.0026953125, + "step": 138 + }, + { + "loss": 0.29100215435028076, + "grad_norm": 6.06061315536499, + "learning_rate": 6.717948717948717e-07, + "num_tokens": 1388124.0, + "completions/mean_length": 661.125, + "completions/min_length": 213.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 408.25, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 14.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44367557764053345, + "rewards/reward_func/std": 0.4120591878890991, + "reward": 0.44367557764053345, + "reward_std": 0.4120591878890991, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020871751476079226, + "sampling/sampling_logp_difference/max": 0.38815033435821533, + "sampling/importance_sampling_ratio/min": 0.7655261754989624, + "sampling/importance_sampling_ratio/mean": 0.9562761187553406, + "sampling/importance_sampling_ratio/max": 1.3044114112854004, + "kl": 0.01922639546683058, + "entropy": 0.056705177994444966, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.7817417755723, + "epoch": 0.00271484375, + "step": 139 + }, + { + "loss": 0.1310022622346878, + "grad_norm": 4.157918930053711, + "learning_rate": 6.692307692307692e-07, + "num_tokens": 1398050.0, + "completions/mean_length": 555.625, + "completions/min_length": 214.0, + "completions/max_length": 890.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 456.5, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.8072420358657837, + "rewards/reward_func/std": 0.15672092139720917, + "reward": 0.8072420358657837, + "reward_std": 0.15672093629837036, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021468987688422203, + "sampling/sampling_logp_difference/max": 0.6287362575531006, + "sampling/importance_sampling_ratio/min": 0.7603280544281006, + "sampling/importance_sampling_ratio/mean": 1.201888084411621, + "sampling/importance_sampling_ratio/max": 2.758666515350342, + "kl": 0.010019546170951799, + "entropy": 0.049071383371483535, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.225352162495255, + "epoch": 0.002734375, + "step": 140 + }, + { + "loss": -0.37337762117385864, + "grad_norm": 2.2201039791107178, + "learning_rate": 6.666666666666666e-07, + "num_tokens": 1407690.0, + "completions/mean_length": 519.5, + "completions/min_length": 262.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 302.6000061035156, + "completions/min_terminated_length": 262.0, + "completions/max_terminated_length": 320.0, + "tools/call_frequency": 12.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6312500238418579, + "rewards/reward_func/std": 0.18449309468269348, + "reward": 0.6312500238418579, + "reward_std": 0.18449309468269348, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003186374669894576, + "sampling/sampling_logp_difference/max": 1.6463193893432617, + "sampling/importance_sampling_ratio/min": 0.1294947862625122, + "sampling/importance_sampling_ratio/mean": 0.9721691608428955, + "sampling/importance_sampling_ratio/max": 2.1127126216888428, + "kl": 0.009739344095578417, + "entropy": 0.06644833640893921, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.636317117139697, + "epoch": 0.00275390625, + "step": 141 + }, + { + "loss": 0.32519054412841797, + "grad_norm": 2.261176109313965, + "learning_rate": 6.64102564102564e-07, + "num_tokens": 1416419.0, + "completions/mean_length": 405.875, + "completions/min_length": 180.0, + "completions/max_length": 908.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 240.0, + "completions/min_terminated_length": 180.0, + "completions/max_terminated_length": 306.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6425378322601318, + "rewards/reward_func/std": 0.2821790874004364, + "reward": 0.6425378322601318, + "reward_std": 0.282179057598114, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037545713130384684, + "sampling/sampling_logp_difference/max": 0.4453563690185547, + "sampling/importance_sampling_ratio/min": 0.4474167823791504, + "sampling/importance_sampling_ratio/mean": 0.7841266989707947, + "sampling/importance_sampling_ratio/max": 1.1552324295043945, + "kl": 0.014098956002271734, + "entropy": 0.07452006003586575, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.338716020807624, + "epoch": 0.0027734375, + "step": 142 + }, + { + "loss": 0.27644386887550354, + "grad_norm": 1.776171088218689, + "learning_rate": 6.615384615384615e-07, + "num_tokens": 1425856.0, + "completions/mean_length": 495.25, + "completions/min_length": 198.0, + "completions/max_length": 942.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 245.1999969482422, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 325.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.535937488079071, + "rewards/reward_func/std": 0.31390509009361267, + "reward": 0.535937488079071, + "reward_std": 0.3139050602912903, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003047117730602622, + "sampling/sampling_logp_difference/max": 0.6487948894500732, + "sampling/importance_sampling_ratio/min": 0.21461831033229828, + "sampling/importance_sampling_ratio/mean": 0.7859681844711304, + "sampling/importance_sampling_ratio/max": 1.3946479558944702, + "kl": 0.01301410143787507, + "entropy": 0.06536604143911973, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.064336232841015, + "epoch": 0.00279296875, + "step": 143 + }, + { + "loss": 0.6302834749221802, + "grad_norm": 3.9180989265441895, + "learning_rate": 6.58974358974359e-07, + "num_tokens": 1434790.0, + "completions/mean_length": 431.125, + "completions/min_length": 230.0, + "completions/max_length": 936.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 359.0000305175781, + "completions/min_terminated_length": 230.0, + "completions/max_terminated_length": 877.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6627367734909058, + "rewards/reward_func/std": 0.17710162699222565, + "reward": 0.6627367734909058, + "reward_std": 0.17710161209106445, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0033962258603423834, + "sampling/sampling_logp_difference/max": 0.4387636184692383, + "sampling/importance_sampling_ratio/min": 0.42081722617149353, + "sampling/importance_sampling_ratio/mean": 1.114455223083496, + "sampling/importance_sampling_ratio/max": 2.0839664936065674, + "kl": 0.011964837845880538, + "entropy": 0.06511250091716647, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.220820628106594, + "epoch": 0.0028125, + "step": 144 + }, + { + "loss": -0.04121624305844307, + "grad_norm": 2.4052810668945312, + "learning_rate": 6.564102564102564e-07, + "num_tokens": 1443650.0, + "completions/mean_length": 421.875, + "completions/min_length": 217.0, + "completions/max_length": 902.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 262.0, + "completions/min_terminated_length": 217.0, + "completions/max_terminated_length": 310.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5845485925674438, + "rewards/reward_func/std": 0.24487519264221191, + "reward": 0.5845485925674438, + "reward_std": 0.24487519264221191, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00477550970390439, + "sampling/sampling_logp_difference/max": 0.6814589500427246, + "sampling/importance_sampling_ratio/min": 0.31506437063217163, + "sampling/importance_sampling_ratio/mean": 0.7406282424926758, + "sampling/importance_sampling_ratio/max": 1.4812383651733398, + "kl": 0.012809886247850955, + "entropy": 0.08077026403043419, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.922136949375272, + "epoch": 0.00283203125, + "step": 145 + }, + { + "loss": 0.4797424376010895, + "grad_norm": 5.2794647216796875, + "learning_rate": 6.538461538461538e-07, + "num_tokens": 1452515.0, + "completions/mean_length": 422.25, + "completions/min_length": 189.0, + "completions/max_length": 892.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 355.14288330078125, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 871.0, + "tools/call_frequency": 10.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5194196701049805, + "rewards/reward_func/std": 0.3028932213783264, + "reward": 0.5194196701049805, + "reward_std": 0.3028932213783264, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004184604622423649, + "sampling/sampling_logp_difference/max": 0.36936235427856445, + "sampling/importance_sampling_ratio/min": 0.5187510251998901, + "sampling/importance_sampling_ratio/mean": 1.2267723083496094, + "sampling/importance_sampling_ratio/max": 1.946203351020813, + "kl": 0.010975092882290483, + "entropy": 0.07590142998378724, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.925903607159853, + "epoch": 0.0028515625, + "step": 146 + }, + { + "loss": 0.2867446541786194, + "grad_norm": 1.3114416599273682, + "learning_rate": 6.512820512820513e-07, + "num_tokens": 1462727.0, + "completions/mean_length": 590.75, + "completions/min_length": 212.0, + "completions/max_length": 955.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 397.3999938964844, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 924.0, + "tools/call_frequency": 13.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4303819537162781, + "rewards/reward_func/std": 0.223826602101326, + "reward": 0.4303819537162781, + "reward_std": 0.2238265872001648, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003363623283803463, + "sampling/sampling_logp_difference/max": 0.808783769607544, + "sampling/importance_sampling_ratio/min": 0.07695215940475464, + "sampling/importance_sampling_ratio/mean": 0.669224739074707, + "sampling/importance_sampling_ratio/max": 1.0303400754928589, + "kl": 0.01025834483152721, + "entropy": 0.06764906679745764, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.36995885707438, + "epoch": 0.00287109375, + "step": 147 + }, + { + "loss": -0.3006729185581207, + "grad_norm": 3.5251197814941406, + "learning_rate": 6.487179487179487e-07, + "num_tokens": 1471530.0, + "completions/mean_length": 414.5, + "completions/min_length": 202.0, + "completions/max_length": 889.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 349.14288330078125, + "completions/min_terminated_length": 202.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5971726179122925, + "rewards/reward_func/std": 0.16380028426647186, + "reward": 0.5971726179122925, + "reward_std": 0.16380028426647186, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037358540575951338, + "sampling/sampling_logp_difference/max": 0.8136688470840454, + "sampling/importance_sampling_ratio/min": 0.2631297707557678, + "sampling/importance_sampling_ratio/mean": 0.9168528914451599, + "sampling/importance_sampling_ratio/max": 1.5997401475906372, + "kl": 0.012448097491869703, + "entropy": 0.09139286691788584, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.172672500833869, + "epoch": 0.002890625, + "step": 148 + }, + { + "loss": 0.4302329123020172, + "grad_norm": 3.691859006881714, + "learning_rate": 6.461538461538462e-07, + "num_tokens": 1480260.0, + "completions/mean_length": 405.625, + "completions/min_length": 213.0, + "completions/max_length": 830.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 345.0000305175781, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 826.0, + "tools/call_frequency": 11.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4779265820980072, + "rewards/reward_func/std": 0.2052510529756546, + "reward": 0.4779265820980072, + "reward_std": 0.2052510529756546, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004294189158827066, + "sampling/sampling_logp_difference/max": 0.7834178805351257, + "sampling/importance_sampling_ratio/min": 0.38934558629989624, + "sampling/importance_sampling_ratio/mean": 0.776461124420166, + "sampling/importance_sampling_ratio/max": 1.3881711959838867, + "kl": 0.011745579482521862, + "entropy": 0.08665771211963147, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.923153538256884, + "epoch": 0.00291015625, + "step": 149 + }, + { + "loss": 0.522196352481842, + "grad_norm": 2.986121416091919, + "learning_rate": 6.435897435897436e-07, + "num_tokens": 1489902.0, + "completions/mean_length": 519.75, + "completions/min_length": 224.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 385.8333435058594, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3920833468437195, + "rewards/reward_func/std": 0.2750439941883087, + "reward": 0.3920833468437195, + "reward_std": 0.2750439941883087, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037516620941460133, + "sampling/sampling_logp_difference/max": 0.675990104675293, + "sampling/importance_sampling_ratio/min": 0.46988779306411743, + "sampling/importance_sampling_ratio/mean": 1.0139318704605103, + "sampling/importance_sampling_ratio/max": 1.6361817121505737, + "kl": 0.010555732354987413, + "entropy": 0.06816002342384309, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.578025391325355, + "epoch": 0.0029296875, + "step": 150 + }, + { + "loss": 0.6153040528297424, + "grad_norm": 3.0264620780944824, + "learning_rate": 6.410256410256411e-07, + "num_tokens": 1498853.0, + "completions/mean_length": 433.625, + "completions/min_length": 196.0, + "completions/max_length": 939.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 269.0, + "completions/min_terminated_length": 196.0, + "completions/max_terminated_length": 307.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5364583730697632, + "rewards/reward_func/std": 0.24114014208316803, + "reward": 0.5364583730697632, + "reward_std": 0.24114012718200684, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003994152415543795, + "sampling/sampling_logp_difference/max": 0.24855375289916992, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9198278784751892, + "sampling/importance_sampling_ratio/max": 1.3371341228485107, + "kl": 0.010235256631858647, + "entropy": 0.0764671117067337, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.885642942041159, + "epoch": 0.00294921875, + "step": 151 + }, + { + "loss": 0.5119287371635437, + "grad_norm": 4.522257328033447, + "learning_rate": 6.384615384615383e-07, + "num_tokens": 1507515.0, + "completions/mean_length": 397.5, + "completions/min_length": 203.0, + "completions/max_length": 902.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 325.4285888671875, + "completions/min_terminated_length": 203.0, + "completions/max_terminated_length": 881.0, + "tools/call_frequency": 9.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5722222328186035, + "rewards/reward_func/std": 0.3031798005104065, + "reward": 0.5722222328186035, + "reward_std": 0.3031798005104065, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003697439096868038, + "sampling/sampling_logp_difference/max": 0.5661635398864746, + "sampling/importance_sampling_ratio/min": 0.6081328988075256, + "sampling/importance_sampling_ratio/mean": 1.176518440246582, + "sampling/importance_sampling_ratio/max": 2.033545732498169, + "kl": 0.013516896578948945, + "entropy": 0.07251896592788398, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.785740850493312, + "epoch": 0.00296875, + "step": 152 + }, + { + "loss": 0.42875751852989197, + "grad_norm": 4.160524368286133, + "learning_rate": 6.358974358974358e-07, + "num_tokens": 1516531.0, + "completions/mean_length": 441.5, + "completions/min_length": 248.0, + "completions/max_length": 971.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 277.3333435058594, + "completions/min_terminated_length": 248.0, + "completions/max_terminated_length": 326.0, + "tools/call_frequency": 9.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5628314018249512, + "rewards/reward_func/std": 0.22927138209342957, + "reward": 0.5628314018249512, + "reward_std": 0.22927138209342957, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004223860800266266, + "sampling/sampling_logp_difference/max": 0.36431002616882324, + "sampling/importance_sampling_ratio/min": 0.7232310175895691, + "sampling/importance_sampling_ratio/mean": 1.2829914093017578, + "sampling/importance_sampling_ratio/max": 2.046522855758667, + "kl": 0.01191656431183219, + "entropy": 0.09031437092926353, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.546109417453408, + "epoch": 0.00298828125, + "step": 153 + }, + { + "loss": 0.8862239718437195, + "grad_norm": 2.9870710372924805, + "learning_rate": 6.333333333333332e-07, + "num_tokens": 1526237.0, + "completions/mean_length": 528.25, + "completions/min_length": 220.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 471.14288330078125, + "completions/min_terminated_length": 220.0, + "completions/max_terminated_length": 923.0, + "tools/call_frequency": 12.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.30260416865348816, + "rewards/reward_func/std": 0.28711530566215515, + "reward": 0.30260416865348816, + "reward_std": 0.28711530566215515, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0029858327470719814, + "sampling/sampling_logp_difference/max": 0.6054059267044067, + "sampling/importance_sampling_ratio/min": 0.19459345936775208, + "sampling/importance_sampling_ratio/mean": 1.2316251993179321, + "sampling/importance_sampling_ratio/max": 2.122978687286377, + "kl": 0.009156704792985693, + "entropy": 0.061896793311461806, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.652526840567589, + "epoch": 0.0030078125, + "step": 154 + }, + { + "loss": 0.23795734345912933, + "grad_norm": 1.8836644887924194, + "learning_rate": 6.307692307692307e-07, + "num_tokens": 1535297.0, + "completions/mean_length": 447.25, + "completions/min_length": 221.0, + "completions/max_length": 955.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 386.5714416503906, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 955.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5796130895614624, + "rewards/reward_func/std": 0.31633225083351135, + "reward": 0.5796130895614624, + "reward_std": 0.31633222103118896, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00439791614189744, + "sampling/sampling_logp_difference/max": 0.5609313249588013, + "sampling/importance_sampling_ratio/min": 0.3560906648635864, + "sampling/importance_sampling_ratio/mean": 0.6424189209938049, + "sampling/importance_sampling_ratio/max": 0.9215748906135559, + "kl": 0.00923186109866947, + "entropy": 0.08094454486854374, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.968086769804358, + "epoch": 0.00302734375, + "step": 155 + }, + { + "loss": 0.3098621368408203, + "grad_norm": 2.603405714035034, + "learning_rate": 6.282051282051281e-07, + "num_tokens": 1544739.0, + "completions/mean_length": 495.625, + "completions/min_length": 216.0, + "completions/max_length": 913.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 438.0000305175781, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4855654835700989, + "rewards/reward_func/std": 0.32695019245147705, + "reward": 0.4855654835700989, + "reward_std": 0.32695022225379944, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026098438538610935, + "sampling/sampling_logp_difference/max": 0.3752455711364746, + "sampling/importance_sampling_ratio/min": 0.5162567496299744, + "sampling/importance_sampling_ratio/mean": 1.1943223476409912, + "sampling/importance_sampling_ratio/max": 2.3033976554870605, + "kl": 0.009112657004152425, + "entropy": 0.06750944338273257, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.554524518549442, + "epoch": 0.003046875, + "step": 156 + }, + { + "loss": 0.5554533004760742, + "grad_norm": 4.1173906326293945, + "learning_rate": 6.256410256410256e-07, + "num_tokens": 1553661.0, + "completions/mean_length": 428.375, + "completions/min_length": 145.0, + "completions/max_length": 951.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 258.0, + "completions/min_terminated_length": 145.0, + "completions/max_terminated_length": 328.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4385416805744171, + "rewards/reward_func/std": 0.4071091115474701, + "reward": 0.4385416805744171, + "reward_std": 0.4071091413497925, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004194855224341154, + "sampling/sampling_logp_difference/max": 0.5675647258758545, + "sampling/importance_sampling_ratio/min": 0.46960145235061646, + "sampling/importance_sampling_ratio/mean": 0.9950351715087891, + "sampling/importance_sampling_ratio/max": 2.133796215057373, + "kl": 0.010622992034768686, + "entropy": 0.07066015258897096, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.637446181848645, + "epoch": 0.00306640625, + "step": 157 + }, + { + "loss": 0.2995336353778839, + "grad_norm": 5.543904781341553, + "learning_rate": 6.23076923076923e-07, + "num_tokens": 1562360.0, + "completions/mean_length": 403.125, + "completions/min_length": 222.0, + "completions/max_length": 901.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 332.0, + "completions/min_terminated_length": 222.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5761440396308899, + "rewards/reward_func/std": 0.3612193763256073, + "reward": 0.5761440396308899, + "reward_std": 0.3612193763256073, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00355580635368824, + "sampling/sampling_logp_difference/max": 0.40850138664245605, + "sampling/importance_sampling_ratio/min": 0.4699437618255615, + "sampling/importance_sampling_ratio/mean": 1.0316890478134155, + "sampling/importance_sampling_ratio/max": 2.1960883140563965, + "kl": 0.01766723592299968, + "entropy": 0.07801524235401303, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.174099631607533, + "epoch": 0.0030859375, + "step": 158 + }, + { + "loss": 0.14627528190612793, + "grad_norm": 2.5553267002105713, + "learning_rate": 6.205128205128205e-07, + "num_tokens": 1571902.0, + "completions/mean_length": 508.375, + "completions/min_length": 199.0, + "completions/max_length": 936.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 258.3999938964844, + "completions/min_terminated_length": 199.0, + "completions/max_terminated_length": 349.0, + "tools/call_frequency": 11.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5427083373069763, + "rewards/reward_func/std": 0.3068476915359497, + "reward": 0.5427083373069763, + "reward_std": 0.3068476915359497, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0036505027674138546, + "sampling/sampling_logp_difference/max": 0.4568955898284912, + "sampling/importance_sampling_ratio/min": 0.4723752439022064, + "sampling/importance_sampling_ratio/mean": 1.2190308570861816, + "sampling/importance_sampling_ratio/max": 2.3727967739105225, + "kl": 0.009929704247042537, + "entropy": 0.063271653954871, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.024187467992306, + "epoch": 0.00310546875, + "step": 159 + }, + { + "loss": 0.25941285490989685, + "grad_norm": 1.8322491645812988, + "learning_rate": 6.179487179487179e-07, + "num_tokens": 1580659.0, + "completions/mean_length": 409.5, + "completions/min_length": 223.0, + "completions/max_length": 958.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 331.14288330078125, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 651.0, + "tools/call_frequency": 9.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5611110925674438, + "rewards/reward_func/std": 0.2291155606508255, + "reward": 0.5611110925674438, + "reward_std": 0.2291155308485031, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004036159720271826, + "sampling/sampling_logp_difference/max": 0.4016055464744568, + "sampling/importance_sampling_ratio/min": 0.5658006072044373, + "sampling/importance_sampling_ratio/mean": 0.9231452941894531, + "sampling/importance_sampling_ratio/max": 1.8791558742523193, + "kl": 0.012470402289181948, + "entropy": 0.09279328328557312, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.57525684311986, + "epoch": 0.003125, + "step": 160 + }, + { + "loss": -0.0181775763630867, + "grad_norm": 8.855108261108398, + "learning_rate": 6.153846153846154e-07, + "num_tokens": 1588733.0, + "completions/mean_length": 325.25, + "completions/min_length": 199.0, + "completions/max_length": 867.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 247.85714721679688, + "completions/min_terminated_length": 199.0, + "completions/max_terminated_length": 352.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5900297164916992, + "rewards/reward_func/std": 0.22045879065990448, + "reward": 0.5900297164916992, + "reward_std": 0.2204587757587433, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005316530354321003, + "sampling/sampling_logp_difference/max": 1.2488718032836914, + "sampling/importance_sampling_ratio/min": 0.21214784681797028, + "sampling/importance_sampling_ratio/mean": 0.981478214263916, + "sampling/importance_sampling_ratio/max": 2.035926103591919, + "kl": 0.01300771755632013, + "entropy": 0.07394251227378845, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.959810141474009, + "epoch": 0.00314453125, + "step": 161 + }, + { + "loss": -0.07461422681808472, + "grad_norm": 6.145508289337158, + "learning_rate": 6.128205128205128e-07, + "num_tokens": 1598356.0, + "completions/mean_length": 517.25, + "completions/min_length": 215.0, + "completions/max_length": 915.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 392.66668701171875, + "completions/min_terminated_length": 215.0, + "completions/max_terminated_length": 881.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3513889014720917, + "rewards/reward_func/std": 0.2690196633338928, + "reward": 0.3513889014720917, + "reward_std": 0.2690196931362152, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037849433720111847, + "sampling/sampling_logp_difference/max": 0.9632071852684021, + "sampling/importance_sampling_ratio/min": 0.35292544960975647, + "sampling/importance_sampling_ratio/mean": 0.9011457562446594, + "sampling/importance_sampling_ratio/max": 1.3546549081802368, + "kl": 0.012988268950721249, + "entropy": 0.08066907722968608, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.70612720400095, + "epoch": 0.0031640625, + "step": 162 + }, + { + "loss": 0.3188670873641968, + "grad_norm": 2.48442006111145, + "learning_rate": 6.102564102564103e-07, + "num_tokens": 1607322.0, + "completions/mean_length": 435.625, + "completions/min_length": 214.0, + "completions/max_length": 945.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 362.8571472167969, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 911.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5816964507102966, + "rewards/reward_func/std": 0.2535242736339569, + "reward": 0.5816964507102966, + "reward_std": 0.2535242736339569, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005086238030344248, + "sampling/sampling_logp_difference/max": 0.4833810329437256, + "sampling/importance_sampling_ratio/min": 0.18676535785198212, + "sampling/importance_sampling_ratio/mean": 0.7554771900177002, + "sampling/importance_sampling_ratio/max": 1.4695371389389038, + "kl": 0.010863532588700764, + "entropy": 0.08865803480148315, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.66091930679977, + "epoch": 0.00318359375, + "step": 163 + }, + { + "loss": 0.34255969524383545, + "grad_norm": 2.6266660690307617, + "learning_rate": 6.076923076923076e-07, + "num_tokens": 1618272.0, + "completions/mean_length": 683.0, + "completions/min_length": 260.0, + "completions/max_length": 958.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 544.2000122070312, + "completions/min_terminated_length": 260.0, + "completions/max_terminated_length": 939.0, + "tools/call_frequency": 15.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.24496528506278992, + "rewards/reward_func/std": 0.24403482675552368, + "reward": 0.24496528506278992, + "reward_std": 0.2440347969532013, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002618669532239437, + "sampling/sampling_logp_difference/max": 0.7108891010284424, + "sampling/importance_sampling_ratio/min": 0.3470892310142517, + "sampling/importance_sampling_ratio/mean": 0.8457489013671875, + "sampling/importance_sampling_ratio/max": 1.4787297248840332, + "kl": 0.010701311657612678, + "entropy": 0.057520760456100106, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.59581253491342, + "epoch": 0.003203125, + "step": 164 + }, + { + "loss": -0.08285842835903168, + "grad_norm": 2.0837743282318115, + "learning_rate": 6.051282051282051e-07, + "num_tokens": 1628535.0, + "completions/mean_length": 596.875, + "completions/min_length": 251.0, + "completions/max_length": 949.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 282.0, + "completions/min_terminated_length": 251.0, + "completions/max_terminated_length": 303.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5807291865348816, + "rewards/reward_func/std": 0.2818055748939514, + "reward": 0.5807291865348816, + "reward_std": 0.2818055748939514, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023227198980748653, + "sampling/sampling_logp_difference/max": 0.6651411056518555, + "sampling/importance_sampling_ratio/min": 0.5650158524513245, + "sampling/importance_sampling_ratio/mean": 1.059631586074829, + "sampling/importance_sampling_ratio/max": 1.6410530805587769, + "kl": 0.007324277074076235, + "entropy": 0.05433546321000904, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.220093557611108, + "epoch": 0.00322265625, + "step": 165 + }, + { + "loss": -0.10450414568185806, + "grad_norm": 2.4499337673187256, + "learning_rate": 6.025641025641025e-07, + "num_tokens": 1637405.0, + "completions/mean_length": 423.875, + "completions/min_length": 219.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 270.8333435058594, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 329.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6112847328186035, + "rewards/reward_func/std": 0.27034035325050354, + "reward": 0.6112847328186035, + "reward_std": 0.27034032344818115, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0036057496909052134, + "sampling/sampling_logp_difference/max": 0.4345517158508301, + "sampling/importance_sampling_ratio/min": 0.23453466594219208, + "sampling/importance_sampling_ratio/mean": 0.8410232067108154, + "sampling/importance_sampling_ratio/max": 1.219643235206604, + "kl": 0.010998349389410578, + "entropy": 0.07986348535632715, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.654726991429925, + "epoch": 0.0032421875, + "step": 166 + }, + { + "loss": 0.1119496300816536, + "grad_norm": 1.6423485279083252, + "learning_rate": 6e-07, + "num_tokens": 1647579.0, + "completions/mean_length": 585.625, + "completions/min_length": 242.0, + "completions/max_length": 982.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 399.6000061035156, + "completions/min_terminated_length": 242.0, + "completions/max_terminated_length": 907.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4482589364051819, + "rewards/reward_func/std": 0.3468341827392578, + "reward": 0.4482589364051819, + "reward_std": 0.3468341827392578, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00238333479501307, + "sampling/sampling_logp_difference/max": 0.5978705883026123, + "sampling/importance_sampling_ratio/min": 0.35502395033836365, + "sampling/importance_sampling_ratio/mean": 0.6912274360656738, + "sampling/importance_sampling_ratio/max": 0.9602075219154358, + "kl": 0.006759845280612353, + "entropy": 0.05811784852994606, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.694384003058076, + "epoch": 0.00326171875, + "step": 167 + }, + { + "loss": 0.304230660200119, + "grad_norm": 2.1441400051116943, + "learning_rate": 5.974358974358974e-07, + "num_tokens": 1659038.0, + "completions/mean_length": 747.75, + "completions/min_length": 247.0, + "completions/max_length": 958.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 460.3333435058594, + "completions/min_terminated_length": 247.0, + "completions/max_terminated_length": 825.0, + "tools/call_frequency": 17.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3454861044883728, + "rewards/reward_func/std": 0.25369811058044434, + "reward": 0.3454861044883728, + "reward_std": 0.25369811058044434, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024844282306730747, + "sampling/sampling_logp_difference/max": 0.9945621490478516, + "sampling/importance_sampling_ratio/min": 0.24796487390995026, + "sampling/importance_sampling_ratio/mean": 0.8720312714576721, + "sampling/importance_sampling_ratio/max": 1.863869309425354, + "kl": 0.005660805938532576, + "entropy": 0.04126555722905323, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.852304438129067, + "epoch": 0.00328125, + "step": 168 + }, + { + "loss": 0.10916894674301147, + "grad_norm": 1.3798094987869263, + "learning_rate": 5.948717948717949e-07, + "num_tokens": 1669151.0, + "completions/mean_length": 579.125, + "completions/min_length": 228.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 537.2857666015625, + "completions/min_terminated_length": 228.0, + "completions/max_terminated_length": 938.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6705695390701294, + "rewards/reward_func/std": 0.27706626057624817, + "reward": 0.6705695390701294, + "reward_std": 0.27706626057624817, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023542148992419243, + "sampling/sampling_logp_difference/max": 0.8026406764984131, + "sampling/importance_sampling_ratio/min": 0.7956541180610657, + "sampling/importance_sampling_ratio/mean": 1.024816632270813, + "sampling/importance_sampling_ratio/max": 1.3976393938064575, + "kl": 0.007480395172024146, + "entropy": 0.04953626141650602, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.036512199789286, + "epoch": 0.00330078125, + "step": 169 + }, + { + "loss": 0.05181247740983963, + "grad_norm": 1.1059091091156006, + "learning_rate": 5.923076923076923e-07, + "num_tokens": 1681757.0, + "completions/mean_length": 890.25, + "completions/min_length": 817.0, + "completions/max_length": 940.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 898.0, + "completions/min_terminated_length": 838.0, + "completions/max_terminated_length": 930.0, + "tools/call_frequency": 21.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.36770832538604736, + "rewards/reward_func/std": 0.36073219776153564, + "reward": 0.36770832538604736, + "reward_std": 0.36073219776153564, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0008356742328032851, + "sampling/sampling_logp_difference/max": 0.4850273132324219, + "sampling/importance_sampling_ratio/min": 0.7696087956428528, + "sampling/importance_sampling_ratio/mean": 0.9061455726623535, + "sampling/importance_sampling_ratio/max": 1.0321497917175293, + "kl": 0.003346894169226289, + "entropy": 0.009310184366768226, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.624124938622117, + "epoch": 0.0033203125, + "step": 170 + }, + { + "loss": -0.21329545974731445, + "grad_norm": 1.862794280052185, + "learning_rate": 5.897435897435898e-07, + "num_tokens": 1692842.0, + "completions/mean_length": 700.375, + "completions/min_length": 161.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 650.8333740234375, + "completions/min_terminated_length": 161.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 17.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5171874761581421, + "rewards/reward_func/std": 0.34578269720077515, + "reward": 0.5171874761581421, + "reward_std": 0.34578269720077515, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001312200096435845, + "sampling/sampling_logp_difference/max": 0.4157141447067261, + "sampling/importance_sampling_ratio/min": 0.6268301010131836, + "sampling/importance_sampling_ratio/mean": 0.9549993872642517, + "sampling/importance_sampling_ratio/max": 1.2631044387817383, + "kl": 0.02412548501160927, + "entropy": 0.03380329086212441, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.311167497187853, + "epoch": 0.00333984375, + "step": 171 + }, + { + "loss": -0.06691871583461761, + "grad_norm": 1.8592655658721924, + "learning_rate": 5.871794871794872e-07, + "num_tokens": 1703612.0, + "completions/mean_length": 661.0, + "completions/min_length": 226.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 589.1666870117188, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 934.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5186756253242493, + "rewards/reward_func/std": 0.21671010553836823, + "reward": 0.5186756253242493, + "reward_std": 0.21671010553836823, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0031000524759292603, + "sampling/sampling_logp_difference/max": 0.581238329410553, + "sampling/importance_sampling_ratio/min": 0.36191749572753906, + "sampling/importance_sampling_ratio/mean": 0.9945606589317322, + "sampling/importance_sampling_ratio/max": 1.7777714729309082, + "kl": 0.00834706169553101, + "entropy": 0.06004905671579763, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.865603243932128, + "epoch": 0.003359375, + "step": 172 + }, + { + "loss": 0.0907532349228859, + "grad_norm": 1.095251202583313, + "learning_rate": 5.846153846153847e-07, + "num_tokens": 1715501.0, + "completions/mean_length": 801.75, + "completions/min_length": 260.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.875, + "completions/mean_terminated_length": 260.0, + "completions/min_terminated_length": 260.0, + "completions/max_terminated_length": 260.0, + "tools/call_frequency": 19.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5927777886390686, + "rewards/reward_func/std": 0.250661164522171, + "reward": 0.5927777886390686, + "reward_std": 0.250661164522171, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013943600933998823, + "sampling/sampling_logp_difference/max": 0.4437302350997925, + "sampling/importance_sampling_ratio/min": 0.6491736769676208, + "sampling/importance_sampling_ratio/mean": 1.2584444284439087, + "sampling/importance_sampling_ratio/max": 2.065054416656494, + "kl": 0.005514308486453956, + "entropy": 0.023820577654987574, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.183460099622607, + "epoch": 0.00337890625, + "step": 173 + }, + { + "loss": -0.07726268470287323, + "grad_norm": 1.2801017761230469, + "learning_rate": 5.82051282051282e-07, + "num_tokens": 1727148.0, + "completions/mean_length": 769.75, + "completions/min_length": 283.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 523.6666870117188, + "completions/min_terminated_length": 283.0, + "completions/max_terminated_length": 929.0, + "tools/call_frequency": 16.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33628469705581665, + "rewards/reward_func/std": 0.21886153519153595, + "reward": 0.33628469705581665, + "reward_std": 0.21886153519153595, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0016883525531738997, + "sampling/sampling_logp_difference/max": 0.4608798027038574, + "sampling/importance_sampling_ratio/min": 0.4600052535533905, + "sampling/importance_sampling_ratio/mean": 1.0773825645446777, + "sampling/importance_sampling_ratio/max": 1.7416183948516846, + "kl": 0.0056008424871833995, + "entropy": 0.035964974435046315, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.563124654814601, + "epoch": 0.0033984375, + "step": 174 + }, + { + "loss": 0.0873384103178978, + "grad_norm": 2.084947109222412, + "learning_rate": 5.794871794871795e-07, + "num_tokens": 1738578.0, + "completions/mean_length": 744.75, + "completions/min_length": 258.0, + "completions/max_length": 964.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 649.7999877929688, + "completions/min_terminated_length": 258.0, + "completions/max_terminated_length": 964.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3880208134651184, + "rewards/reward_func/std": 0.3230208158493042, + "reward": 0.3880208134651184, + "reward_std": 0.3230208158493042, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0014924919232726097, + "sampling/sampling_logp_difference/max": 0.5461215972900391, + "sampling/importance_sampling_ratio/min": 0.5948064923286438, + "sampling/importance_sampling_ratio/mean": 1.0687581300735474, + "sampling/importance_sampling_ratio/max": 1.8161176443099976, + "kl": 0.005872359950444661, + "entropy": 0.027467914653243497, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.39698240160942, + "epoch": 0.00341796875, + "step": 175 + }, + { + "loss": 0.12764331698417664, + "grad_norm": 1.5614897012710571, + "learning_rate": 5.769230769230768e-07, + "num_tokens": 1751358.0, + "completions/mean_length": 911.625, + "completions/min_length": 870.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 910.0, + "completions/min_terminated_length": 893.0, + "completions/max_terminated_length": 927.0, + "tools/call_frequency": 20.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.2984375059604645, + "rewards/reward_func/std": 0.3077208995819092, + "reward": 0.2984375059604645, + "reward_std": 0.3077208697795868, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013841277686879039, + "sampling/sampling_logp_difference/max": 0.8849995136260986, + "sampling/importance_sampling_ratio/min": 0.409952312707901, + "sampling/importance_sampling_ratio/mean": 1.1049787998199463, + "sampling/importance_sampling_ratio/max": 2.174083709716797, + "kl": 0.003968927223468199, + "entropy": 0.01302684290567413, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.884879499673843, + "epoch": 0.0034375, + "step": 176 + }, + { + "loss": 0.13744449615478516, + "grad_norm": 1.5396581888198853, + "learning_rate": 5.743589743589743e-07, + "num_tokens": 1763265.0, + "completions/mean_length": 803.75, + "completions/min_length": 237.0, + "completions/max_length": 914.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 726.0, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 914.0, + "tools/call_frequency": 18.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4007575511932373, + "rewards/reward_func/std": 0.2914959192276001, + "reward": 0.4007575511932373, + "reward_std": 0.2914959192276001, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013788768555969, + "sampling/sampling_logp_difference/max": 0.5609216690063477, + "sampling/importance_sampling_ratio/min": 0.4968473017215729, + "sampling/importance_sampling_ratio/mean": 0.9652271270751953, + "sampling/importance_sampling_ratio/max": 1.875111699104309, + "kl": 0.006790020925109275, + "entropy": 0.026314845425076783, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.54900706000626, + "epoch": 0.00345703125, + "step": 177 + }, + { + "loss": 0.1776115596294403, + "grad_norm": 1.5436739921569824, + "learning_rate": 5.717948717948717e-07, + "num_tokens": 1775336.0, + "completions/mean_length": 824.875, + "completions/min_length": 309.0, + "completions/max_length": 982.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 674.0, + "completions/min_terminated_length": 309.0, + "completions/max_terminated_length": 860.0, + "tools/call_frequency": 19.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4862847328186035, + "rewards/reward_func/std": 0.34214577078819275, + "reward": 0.4862847328186035, + "reward_std": 0.34214577078819275, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0012098945444449782, + "sampling/sampling_logp_difference/max": 0.40104949474334717, + "sampling/importance_sampling_ratio/min": 0.5041077136993408, + "sampling/importance_sampling_ratio/mean": 1.0819485187530518, + "sampling/importance_sampling_ratio/max": 2.2338321208953857, + "kl": 0.004166310995060485, + "entropy": 0.023906336311483756, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.861862063407898, + "epoch": 0.0034765625, + "step": 178 + }, + { + "loss": 0.06973915547132492, + "grad_norm": 2.3743228912353516, + "learning_rate": 5.692307692307692e-07, + "num_tokens": 1786571.0, + "completions/mean_length": 719.25, + "completions/min_length": 172.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 443.0, + "completions/min_terminated_length": 172.0, + "completions/max_terminated_length": 930.0, + "tools/call_frequency": 17.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5941964387893677, + "rewards/reward_func/std": 0.41950416564941406, + "reward": 0.5941964387893677, + "reward_std": 0.41950416564941406, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0010488891275599599, + "sampling/sampling_logp_difference/max": 0.5246936082839966, + "sampling/importance_sampling_ratio/min": 0.60358726978302, + "sampling/importance_sampling_ratio/mean": 0.9112227559089661, + "sampling/importance_sampling_ratio/max": 1.6719774007797241, + "kl": 0.012274848792003468, + "entropy": 0.021412082627648488, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.71326862461865, + "epoch": 0.00349609375, + "step": 179 + }, + { + "loss": 0.7414481043815613, + "grad_norm": 3.6853671073913574, + "learning_rate": 5.666666666666666e-07, + "num_tokens": 1797261.0, + "completions/mean_length": 651.0, + "completions/min_length": 255.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 433.75, + "completions/min_terminated_length": 255.0, + "completions/max_terminated_length": 894.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6178819537162781, + "rewards/reward_func/std": 0.2674100995063782, + "reward": 0.6178819537162781, + "reward_std": 0.26741012930870056, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00226965406909585, + "sampling/sampling_logp_difference/max": 0.4075760841369629, + "sampling/importance_sampling_ratio/min": 0.597588300704956, + "sampling/importance_sampling_ratio/mean": 1.1169991493225098, + "sampling/importance_sampling_ratio/max": 2.3260843753814697, + "kl": 0.007230634582811035, + "entropy": 0.044300637615378946, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.704331943765283, + "epoch": 0.003515625, + "step": 180 + }, + { + "loss": -0.0398632287979126, + "grad_norm": 1.7654871940612793, + "learning_rate": 5.641025641025641e-07, + "num_tokens": 1808671.0, + "completions/mean_length": 740.75, + "completions/min_length": 241.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 645.0, + "completions/min_terminated_length": 241.0, + "completions/max_terminated_length": 921.0, + "tools/call_frequency": 16.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5028409361839294, + "rewards/reward_func/std": 0.30690863728523254, + "reward": 0.5028409361839294, + "reward_std": 0.30690863728523254, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001978930551558733, + "sampling/sampling_logp_difference/max": 0.48525261878967285, + "sampling/importance_sampling_ratio/min": 0.508034348487854, + "sampling/importance_sampling_ratio/mean": 0.9266951084136963, + "sampling/importance_sampling_ratio/max": 1.1062053442001343, + "kl": 0.010439296762342565, + "entropy": 0.04474894353188574, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.732753520831466, + "epoch": 0.00353515625, + "step": 181 + }, + { + "loss": 0.048170968890190125, + "grad_norm": 1.8770523071289062, + "learning_rate": 5.615384615384615e-07, + "num_tokens": 1820671.0, + "completions/mean_length": 815.625, + "completions/min_length": 235.0, + "completions/max_length": 955.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 575.0, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 915.0, + "tools/call_frequency": 18.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3990909159183502, + "rewards/reward_func/std": 0.26890167593955994, + "reward": 0.3990909159183502, + "reward_std": 0.26890167593955994, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013824626803398132, + "sampling/sampling_logp_difference/max": 0.40310096740722656, + "sampling/importance_sampling_ratio/min": 0.7782444953918457, + "sampling/importance_sampling_ratio/mean": 1.1628082990646362, + "sampling/importance_sampling_ratio/max": 1.6977664232254028, + "kl": 0.008859947338351049, + "entropy": 0.02060773695120588, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.770052909851074, + "epoch": 0.0035546875, + "step": 182 + }, + { + "loss": -0.012779004871845245, + "grad_norm": 1.0744664669036865, + "learning_rate": 5.58974358974359e-07, + "num_tokens": 1832094.0, + "completions/mean_length": 743.0, + "completions/min_length": 249.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 639.7999877929688, + "completions/min_terminated_length": 249.0, + "completions/max_terminated_length": 911.0, + "tools/call_frequency": 17.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4671996831893921, + "rewards/reward_func/std": 0.3785083591938019, + "reward": 0.4671996831893921, + "reward_std": 0.3785083591938019, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00168430688790977, + "sampling/sampling_logp_difference/max": 0.4883723258972168, + "sampling/importance_sampling_ratio/min": 0.5030224323272705, + "sampling/importance_sampling_ratio/mean": 0.8352503180503845, + "sampling/importance_sampling_ratio/max": 1.8873846530914307, + "kl": 0.0081527239526622, + "entropy": 0.04421579372137785, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.03845264390111, + "epoch": 0.00357421875, + "step": 183 + }, + { + "loss": -0.0632232129573822, + "grad_norm": 1.4021724462509155, + "learning_rate": 5.564102564102564e-07, + "num_tokens": 1842850.0, + "completions/mean_length": 659.25, + "completions/min_length": 184.0, + "completions/max_length": 913.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 271.66668701171875, + "completions/min_terminated_length": 184.0, + "completions/max_terminated_length": 321.0, + "tools/call_frequency": 15.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44236111640930176, + "rewards/reward_func/std": 0.3260877728462219, + "reward": 0.44236111640930176, + "reward_std": 0.3260877728462219, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017378672491759062, + "sampling/sampling_logp_difference/max": 0.3742084503173828, + "sampling/importance_sampling_ratio/min": 0.46653109788894653, + "sampling/importance_sampling_ratio/mean": 0.8386024832725525, + "sampling/importance_sampling_ratio/max": 1.5258013010025024, + "kl": 0.009343032870674506, + "entropy": 0.04418362941942178, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.049634870141745, + "epoch": 0.00359375, + "step": 184 + }, + { + "loss": 0.061900924891233444, + "grad_norm": 2.0680031776428223, + "learning_rate": 5.538461538461539e-07, + "num_tokens": 1853801.0, + "completions/mean_length": 683.625, + "completions/min_length": 224.0, + "completions/max_length": 924.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 480.75, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 924.0, + "tools/call_frequency": 15.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5740165114402771, + "rewards/reward_func/std": 0.3226163387298584, + "reward": 0.5740165114402771, + "reward_std": 0.322616308927536, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023342303466051817, + "sampling/sampling_logp_difference/max": 2.247776985168457, + "sampling/importance_sampling_ratio/min": 0.1233975812792778, + "sampling/importance_sampling_ratio/mean": 0.7472628951072693, + "sampling/importance_sampling_ratio/max": 1.1330071687698364, + "kl": 0.008402389008551836, + "entropy": 0.041429692500969395, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.383073119446635, + "epoch": 0.00361328125, + "step": 185 + }, + { + "loss": 0.042558494955301285, + "grad_norm": 4.14876127243042, + "learning_rate": 5.512820512820513e-07, + "num_tokens": 1865798.0, + "completions/mean_length": 812.75, + "completions/min_length": 383.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 690.6666870117188, + "completions/min_terminated_length": 383.0, + "completions/max_terminated_length": 846.0, + "tools/call_frequency": 18.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5248264074325562, + "rewards/reward_func/std": 0.26732560992240906, + "reward": 0.5248264074325562, + "reward_std": 0.26732560992240906, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021699636708945036, + "sampling/sampling_logp_difference/max": 2.631688117980957, + "sampling/importance_sampling_ratio/min": 0.0400875061750412, + "sampling/importance_sampling_ratio/mean": 0.7283908128738403, + "sampling/importance_sampling_ratio/max": 1.2373052835464478, + "kl": 0.01028957508970052, + "entropy": 0.025767216400709003, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.361302768811584, + "epoch": 0.0036328125, + "step": 186 + }, + { + "loss": 0.2302151322364807, + "grad_norm": 2.195815324783325, + "learning_rate": 5.487179487179488e-07, + "num_tokens": 1876116.0, + "completions/mean_length": 604.5, + "completions/min_length": 242.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 410.8000183105469, + "completions/min_terminated_length": 242.0, + "completions/max_terminated_length": 939.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.34843748807907104, + "rewards/reward_func/std": 0.3738337457180023, + "reward": 0.34843748807907104, + "reward_std": 0.37383371591567993, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003357765031978488, + "sampling/sampling_logp_difference/max": 0.9910693168640137, + "sampling/importance_sampling_ratio/min": 0.3192046582698822, + "sampling/importance_sampling_ratio/mean": 0.8746747970581055, + "sampling/importance_sampling_ratio/max": 1.4263755083084106, + "kl": 0.007467495102901012, + "entropy": 0.06460548669565469, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.681714721024036, + "epoch": 0.00365234375, + "step": 187 + }, + { + "loss": 0.17691919207572937, + "grad_norm": 1.2970560789108276, + "learning_rate": 5.461538461538461e-07, + "num_tokens": 1887602.0, + "completions/mean_length": 750.0, + "completions/min_length": 294.0, + "completions/max_length": 991.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 626.5, + "completions/min_terminated_length": 294.0, + "completions/max_terminated_length": 991.0, + "tools/call_frequency": 17.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5989583730697632, + "rewards/reward_func/std": 0.2760509252548218, + "reward": 0.5989583730697632, + "reward_std": 0.2760509252548218, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023538502864539623, + "sampling/sampling_logp_difference/max": 1.29571533203125, + "sampling/importance_sampling_ratio/min": 0.13112270832061768, + "sampling/importance_sampling_ratio/mean": 0.7477844953536987, + "sampling/importance_sampling_ratio/max": 1.2435250282287598, + "kl": 0.0067175611038692296, + "entropy": 0.03832696593599394, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.578387018293142, + "epoch": 0.003671875, + "step": 188 + }, + { + "loss": 0.5444338321685791, + "grad_norm": 1.5994304418563843, + "learning_rate": 5.435897435897435e-07, + "num_tokens": 1897766.0, + "completions/mean_length": 584.25, + "completions/min_length": 248.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 387.20001220703125, + "completions/min_terminated_length": 248.0, + "completions/max_terminated_length": 858.0, + "tools/call_frequency": 13.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5570684671401978, + "rewards/reward_func/std": 0.23826342821121216, + "reward": 0.5570684671401978, + "reward_std": 0.23826342821121216, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028488917741924524, + "sampling/sampling_logp_difference/max": 0.42786920070648193, + "sampling/importance_sampling_ratio/min": 0.5314720869064331, + "sampling/importance_sampling_ratio/mean": 1.0710723400115967, + "sampling/importance_sampling_ratio/max": 1.759606957435608, + "kl": 0.010381265339674428, + "entropy": 0.059635387267917395, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.211228869855404, + "epoch": 0.00369140625, + "step": 189 + }, + { + "loss": -0.0018901117146015167, + "grad_norm": 2.289222478866577, + "learning_rate": 5.41025641025641e-07, + "num_tokens": 1907444.0, + "completions/mean_length": 524.375, + "completions/min_length": 263.0, + "completions/max_length": 931.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 397.5, + "completions/min_terminated_length": 263.0, + "completions/max_terminated_length": 874.0, + "tools/call_frequency": 12.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4506944417953491, + "rewards/reward_func/std": 0.2627513110637665, + "reward": 0.4506944417953491, + "reward_std": 0.2627513110637665, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003237991826608777, + "sampling/sampling_logp_difference/max": 0.4367094039916992, + "sampling/importance_sampling_ratio/min": 0.6466103792190552, + "sampling/importance_sampling_ratio/mean": 1.1330586671829224, + "sampling/importance_sampling_ratio/max": 1.6242953538894653, + "kl": 0.008570382284233347, + "entropy": 0.06200175161939114, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.974699459969997, + "epoch": 0.0037109375, + "step": 190 + }, + { + "loss": 0.12589320540428162, + "grad_norm": 1.0693005323410034, + "learning_rate": 5.384615384615384e-07, + "num_tokens": 1918357.0, + "completions/mean_length": 678.5, + "completions/min_length": 245.0, + "completions/max_length": 981.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 529.2000122070312, + "completions/min_terminated_length": 245.0, + "completions/max_terminated_length": 928.0, + "tools/call_frequency": 15.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35096728801727295, + "rewards/reward_func/std": 0.3466663658618927, + "reward": 0.35096728801727295, + "reward_std": 0.3466663658618927, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001775024109520018, + "sampling/sampling_logp_difference/max": 0.6201142072677612, + "sampling/importance_sampling_ratio/min": 0.34663525223731995, + "sampling/importance_sampling_ratio/mean": 0.7487412095069885, + "sampling/importance_sampling_ratio/max": 0.9951271414756775, + "kl": 0.0071008751838235185, + "entropy": 0.03304017678601667, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.366790693253279, + "epoch": 0.00373046875, + "step": 191 + }, + { + "loss": -0.2735997438430786, + "grad_norm": 5.183953762054443, + "learning_rate": 5.358974358974359e-07, + "num_tokens": 1926707.0, + "completions/mean_length": 356.625, + "completions/min_length": 226.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 279.5714416503906, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 382.0, + "tools/call_frequency": 8.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47842487692832947, + "rewards/reward_func/std": 0.2547184228897095, + "reward": 0.47842487692832947, + "reward_std": 0.2547183930873871, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006202317774295807, + "sampling/sampling_logp_difference/max": 0.6573474407196045, + "sampling/importance_sampling_ratio/min": 0.39685243368148804, + "sampling/importance_sampling_ratio/mean": 1.1366976499557495, + "sampling/importance_sampling_ratio/max": 2.7703399658203125, + "kl": 0.015427531441673636, + "entropy": 0.10555417544674128, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.518719516694546, + "epoch": 0.00375, + "step": 192 + }, + { + "loss": -0.13718733191490173, + "grad_norm": 6.1737494468688965, + "learning_rate": 5.333333333333333e-07, + "num_tokens": 1934243.0, + "completions/mean_length": 258.25, + "completions/min_length": 181.0, + "completions/max_length": 352.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 258.25, + "completions/min_terminated_length": 181.0, + "completions/max_terminated_length": 352.0, + "tools/call_frequency": 6.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5272916555404663, + "rewards/reward_func/std": 0.33791252970695496, + "reward": 0.5272916555404663, + "reward_std": 0.33791255950927734, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.007527089212089777, + "sampling/sampling_logp_difference/max": 0.3819364309310913, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9846656322479248, + "sampling/importance_sampling_ratio/max": 2.2012975215911865, + "kl": 0.017605706700123847, + "entropy": 0.09424046706408262, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.858725691214204, + "epoch": 0.00376953125, + "step": 193 + }, + { + "loss": 0.5524022579193115, + "grad_norm": 2.3989739418029785, + "learning_rate": 5.307692307692308e-07, + "num_tokens": 1943369.0, + "completions/mean_length": 454.125, + "completions/min_length": 212.0, + "completions/max_length": 960.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 291.0, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 328.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.38333335518836975, + "rewards/reward_func/std": 0.28720223903656006, + "reward": 0.38333335518836975, + "reward_std": 0.28720223903656006, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00428895466029644, + "sampling/sampling_logp_difference/max": 0.39358532428741455, + "sampling/importance_sampling_ratio/min": 0.29298388957977295, + "sampling/importance_sampling_ratio/mean": 0.8569666147232056, + "sampling/importance_sampling_ratio/max": 1.2491344213485718, + "kl": 0.00953130010748282, + "entropy": 0.07793420844245702, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.815865341573954, + "epoch": 0.0037890625, + "step": 194 + }, + { + "loss": 0.28747427463531494, + "grad_norm": 2.096708297729492, + "learning_rate": 5.282051282051282e-07, + "num_tokens": 1952358.0, + "completions/mean_length": 438.875, + "completions/min_length": 270.0, + "completions/max_length": 909.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 371.71429443359375, + "completions/min_terminated_length": 270.0, + "completions/max_terminated_length": 866.0, + "tools/call_frequency": 10.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3288194537162781, + "rewards/reward_func/std": 0.2906944751739502, + "reward": 0.3288194537162781, + "reward_std": 0.2906944751739502, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004290646407753229, + "sampling/sampling_logp_difference/max": 0.4760730266571045, + "sampling/importance_sampling_ratio/min": 0.19146046042442322, + "sampling/importance_sampling_ratio/mean": 0.7260392904281616, + "sampling/importance_sampling_ratio/max": 1.7244185209274292, + "kl": 0.014946554088965058, + "entropy": 0.08750252329627983, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.191427651792765, + "epoch": 0.00380859375, + "step": 195 + }, + { + "loss": -0.05659966543316841, + "grad_norm": 2.811668634414673, + "learning_rate": 5.256410256410256e-07, + "num_tokens": 1960328.0, + "completions/mean_length": 311.375, + "completions/min_length": 46.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 224.2857208251953, + "completions/min_terminated_length": 46.0, + "completions/max_terminated_length": 291.0, + "tools/call_frequency": 7.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5241950750350952, + "rewards/reward_func/std": 0.32794585824012756, + "reward": 0.5241950750350952, + "reward_std": 0.3279458284378052, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004260271321982145, + "sampling/sampling_logp_difference/max": 0.40829408168792725, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.8962926864624023, + "sampling/importance_sampling_ratio/max": 1.9560596942901611, + "kl": 0.010059734213427873, + "entropy": 0.07946439948864281, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.008460940793157, + "epoch": 0.003828125, + "step": 196 + }, + { + "loss": 0.3523959219455719, + "grad_norm": 3.071737051010132, + "learning_rate": 5.23076923076923e-07, + "num_tokens": 1969543.0, + "completions/mean_length": 467.625, + "completions/min_length": 196.0, + "completions/max_length": 901.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 407.8571472167969, + "completions/min_terminated_length": 196.0, + "completions/max_terminated_length": 901.0, + "tools/call_frequency": 11.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6577380895614624, + "rewards/reward_func/std": 0.27338993549346924, + "reward": 0.6577380895614624, + "reward_std": 0.27338993549346924, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003225533990189433, + "sampling/sampling_logp_difference/max": 0.5515315532684326, + "sampling/importance_sampling_ratio/min": 0.6030044555664062, + "sampling/importance_sampling_ratio/mean": 1.5128161907196045, + "sampling/importance_sampling_ratio/max": 2.7110981941223145, + "kl": 0.009788234950974584, + "entropy": 0.07312169903889298, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.090986080467701, + "epoch": 0.00384765625, + "step": 197 + }, + { + "loss": -0.1982431411743164, + "grad_norm": 4.271006107330322, + "learning_rate": 5.205128205128205e-07, + "num_tokens": 1978572.0, + "completions/mean_length": 443.125, + "completions/min_length": 237.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 288.16668701171875, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 362.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.49289458990097046, + "rewards/reward_func/std": 0.2912410497665405, + "reward": 0.49289458990097046, + "reward_std": 0.2912410497665405, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004001446533948183, + "sampling/sampling_logp_difference/max": 0.44869112968444824, + "sampling/importance_sampling_ratio/min": 0.5218845009803772, + "sampling/importance_sampling_ratio/mean": 1.0164332389831543, + "sampling/importance_sampling_ratio/max": 1.5761735439300537, + "kl": 0.010853642830625176, + "entropy": 0.07734013628214598, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.62918847054243, + "epoch": 0.0038671875, + "step": 198 + }, + { + "loss": -0.4780837297439575, + "grad_norm": 3.400224208831787, + "learning_rate": 5.179487179487179e-07, + "num_tokens": 1987754.0, + "completions/mean_length": 461.875, + "completions/min_length": 266.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 394.71429443359375, + "completions/min_terminated_length": 266.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.42115575075149536, + "rewards/reward_func/std": 0.1305294930934906, + "reward": 0.42115575075149536, + "reward_std": 0.1305294781923294, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006060873623937368, + "sampling/sampling_logp_difference/max": 0.7177753448486328, + "sampling/importance_sampling_ratio/min": 0.21639855206012726, + "sampling/importance_sampling_ratio/mean": 0.8621686697006226, + "sampling/importance_sampling_ratio/max": 1.9758408069610596, + "kl": 0.01434451574459672, + "entropy": 0.09645413747057319, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.839237881824374, + "epoch": 0.00388671875, + "step": 199 + }, + { + "loss": 0.338604211807251, + "grad_norm": 3.6931207180023193, + "learning_rate": 5.153846153846153e-07, + "num_tokens": 1996570.0, + "completions/mean_length": 417.625, + "completions/min_length": 197.0, + "completions/max_length": 939.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 244.1666717529297, + "completions/min_terminated_length": 197.0, + "completions/max_terminated_length": 297.0, + "tools/call_frequency": 9.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4528124928474426, + "rewards/reward_func/std": 0.2584250569343567, + "reward": 0.4528124928474426, + "reward_std": 0.2584250569343567, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004665685817599297, + "sampling/sampling_logp_difference/max": 0.9998749494552612, + "sampling/importance_sampling_ratio/min": 0.17221307754516602, + "sampling/importance_sampling_ratio/mean": 0.6659213900566101, + "sampling/importance_sampling_ratio/max": 1.481041669845581, + "kl": 0.017896257719257846, + "entropy": 0.07390063151251525, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.589987136423588, + "epoch": 0.00390625, + "step": 200 + }, + { + "loss": 0.3242771029472351, + "grad_norm": 3.5446810722351074, + "learning_rate": 5.128205128205127e-07, + "num_tokens": 2006102.0, + "completions/mean_length": 505.0, + "completions/min_length": 237.0, + "completions/max_length": 906.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 279.20001220703125, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 313.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47968751192092896, + "rewards/reward_func/std": 0.22340081632137299, + "reward": 0.47968751192092896, + "reward_std": 0.22340081632137299, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003684370080009103, + "sampling/sampling_logp_difference/max": 0.2943863868713379, + "sampling/importance_sampling_ratio/min": 0.4936775863170624, + "sampling/importance_sampling_ratio/mean": 0.9689271450042725, + "sampling/importance_sampling_ratio/max": 1.3549470901489258, + "kl": 0.014853379048872739, + "entropy": 0.080339947482571, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.423355439677835, + "epoch": 0.00392578125, + "step": 201 + }, + { + "loss": 0.1463557928800583, + "grad_norm": 2.1024553775787354, + "learning_rate": 5.102564102564102e-07, + "num_tokens": 2016160.0, + "completions/mean_length": 572.375, + "completions/min_length": 237.0, + "completions/max_length": 914.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 462.66668701171875, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 880.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5309027433395386, + "rewards/reward_func/std": 0.22160226106643677, + "reward": 0.5309027433395386, + "reward_std": 0.22160223126411438, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002581182634457946, + "sampling/sampling_logp_difference/max": 0.701019287109375, + "sampling/importance_sampling_ratio/min": 0.27553728222846985, + "sampling/importance_sampling_ratio/mean": 1.034733772277832, + "sampling/importance_sampling_ratio/max": 1.9526845216751099, + "kl": 0.01041742751840502, + "entropy": 0.04882960766553879, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.479490241035819, + "epoch": 0.0039453125, + "step": 202 + }, + { + "loss": 0.9251823425292969, + "grad_norm": 5.7648797035217285, + "learning_rate": 5.076923076923076e-07, + "num_tokens": 2024585.0, + "completions/mean_length": 367.375, + "completions/min_length": 185.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 291.5714416503906, + "completions/min_terminated_length": 185.0, + "completions/max_terminated_length": 348.0, + "tools/call_frequency": 8.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.46629464626312256, + "rewards/reward_func/std": 0.21852059662342072, + "reward": 0.46629464626312256, + "reward_std": 0.21852059662342072, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005003891885280609, + "sampling/sampling_logp_difference/max": 0.7454648017883301, + "sampling/importance_sampling_ratio/min": 0.3062371611595154, + "sampling/importance_sampling_ratio/mean": 1.0065243244171143, + "sampling/importance_sampling_ratio/max": 1.665789008140564, + "kl": 0.010844340155017562, + "entropy": 0.07929291296750307, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.46343102119863, + "epoch": 0.00396484375, + "step": 203 + }, + { + "loss": -0.1101216971874237, + "grad_norm": 4.502627372741699, + "learning_rate": 5.051282051282051e-07, + "num_tokens": 2032644.0, + "completions/mean_length": 323.375, + "completions/min_length": 208.0, + "completions/max_length": 885.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 243.1428680419922, + "completions/min_terminated_length": 208.0, + "completions/max_terminated_length": 342.0, + "tools/call_frequency": 7.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6764137148857117, + "rewards/reward_func/std": 0.2366914451122284, + "reward": 0.6764137148857117, + "reward_std": 0.236691415309906, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0052929953671991825, + "sampling/sampling_logp_difference/max": 0.49173223972320557, + "sampling/importance_sampling_ratio/min": 0.6080746054649353, + "sampling/importance_sampling_ratio/mean": 0.9244847297668457, + "sampling/importance_sampling_ratio/max": 1.5608410835266113, + "kl": 0.01001882302807644, + "entropy": 0.08997270854888484, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.066620245575905, + "epoch": 0.003984375, + "step": 204 + }, + { + "loss": 0.3970499336719513, + "grad_norm": 3.1446692943573, + "learning_rate": 5.025641025641025e-07, + "num_tokens": 2041986.0, + "completions/mean_length": 482.875, + "completions/min_length": 173.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 351.0, + "completions/min_terminated_length": 173.0, + "completions/max_terminated_length": 871.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5615575313568115, + "rewards/reward_func/std": 0.266404390335083, + "reward": 0.5615575313568115, + "reward_std": 0.266404390335083, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028683156706392765, + "sampling/sampling_logp_difference/max": 0.4733877182006836, + "sampling/importance_sampling_ratio/min": 0.4570561647415161, + "sampling/importance_sampling_ratio/mean": 0.84371018409729, + "sampling/importance_sampling_ratio/max": 1.331742763519287, + "kl": 0.015268344664946198, + "entropy": 0.061559128924272954, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.3717425391078, + "epoch": 0.00400390625, + "step": 205 + }, + { + "loss": 0.13809508085250854, + "grad_norm": 4.212330341339111, + "learning_rate": 5e-07, + "num_tokens": 2049517.0, + "completions/mean_length": 255.5, + "completions/min_length": 207.0, + "completions/max_length": 369.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 255.5, + "completions/min_terminated_length": 207.0, + "completions/max_terminated_length": 369.0, + "tools/call_frequency": 6.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6185605525970459, + "rewards/reward_func/std": 0.31693756580352783, + "reward": 0.6185605525970459, + "reward_std": 0.31693753600120544, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.007355873938649893, + "sampling/sampling_logp_difference/max": 0.4207613468170166, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.8174629211425781, + "sampling/importance_sampling_ratio/max": 1.3213638067245483, + "kl": 0.02094058832153678, + "entropy": 0.0903808488510549, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.563189772889018, + "epoch": 0.0040234375, + "step": 206 + }, + { + "loss": -0.0808074101805687, + "grad_norm": 2.4486820697784424, + "learning_rate": 4.974358974358974e-07, + "num_tokens": 2059741.0, + "completions/mean_length": 592.625, + "completions/min_length": 246.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 410.0, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 917.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5107070803642273, + "rewards/reward_func/std": 0.32080280780792236, + "reward": 0.5107070803642273, + "reward_std": 0.32080280780792236, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002765382407233119, + "sampling/sampling_logp_difference/max": 0.7733519077301025, + "sampling/importance_sampling_ratio/min": 0.24144743382930756, + "sampling/importance_sampling_ratio/mean": 0.7263538837432861, + "sampling/importance_sampling_ratio/max": 1.0570627450942993, + "kl": 0.008400701044593006, + "entropy": 0.07065520447213203, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.374073598533869, + "epoch": 0.00404296875, + "step": 207 + }, + { + "loss": 0.3223092257976532, + "grad_norm": 3.047156572341919, + "learning_rate": 4.948717948717949e-07, + "num_tokens": 2069937.0, + "completions/mean_length": 589.125, + "completions/min_length": 216.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 388.0, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 867.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.32430556416511536, + "rewards/reward_func/std": 0.2749047875404358, + "reward": 0.32430556416511536, + "reward_std": 0.2749047577381134, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024868687614798546, + "sampling/sampling_logp_difference/max": 0.3871268033981323, + "sampling/importance_sampling_ratio/min": 0.12055175006389618, + "sampling/importance_sampling_ratio/mean": 0.9571755528450012, + "sampling/importance_sampling_ratio/max": 1.6052085161209106, + "kl": 0.0077249286259757355, + "entropy": 0.05428423878038302, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.64506302587688, + "epoch": 0.0040625, + "step": 208 + }, + { + "loss": 0.8125616312026978, + "grad_norm": 4.892197608947754, + "learning_rate": 4.923076923076923e-07, + "num_tokens": 2078165.0, + "completions/mean_length": 344.0, + "completions/min_length": 223.0, + "completions/max_length": 903.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 264.14288330078125, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 340.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4633680582046509, + "rewards/reward_func/std": 0.23341591656208038, + "reward": 0.4633680582046509, + "reward_std": 0.23341591656208038, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005358301103115082, + "sampling/sampling_logp_difference/max": 0.5992922782897949, + "sampling/importance_sampling_ratio/min": 0.5272842049598694, + "sampling/importance_sampling_ratio/mean": 0.9490111470222473, + "sampling/importance_sampling_ratio/max": 1.706559658050537, + "kl": 0.011027131287846714, + "entropy": 0.08717303723096848, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.32727325707674, + "epoch": 0.00408203125, + "step": 209 + }, + { + "loss": 0.87088942527771, + "grad_norm": 3.682572364807129, + "learning_rate": 4.897435897435897e-07, + "num_tokens": 2087158.0, + "completions/mean_length": 438.5, + "completions/min_length": 232.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 274.3333435058594, + "completions/min_terminated_length": 232.0, + "completions/max_terminated_length": 318.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4682638943195343, + "rewards/reward_func/std": 0.2810515761375427, + "reward": 0.4682638943195343, + "reward_std": 0.28105154633522034, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004829747136682272, + "sampling/sampling_logp_difference/max": 0.4401230812072754, + "sampling/importance_sampling_ratio/min": 0.3008347153663635, + "sampling/importance_sampling_ratio/mean": 1.0021939277648926, + "sampling/importance_sampling_ratio/max": 1.8500750064849854, + "kl": 0.01295346228289418, + "entropy": 0.0967433350160718, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.559614906087518, + "epoch": 0.0041015625, + "step": 210 + }, + { + "loss": 0.2723321318626404, + "grad_norm": 2.6060197353363037, + "learning_rate": 4.871794871794871e-07, + "num_tokens": 2096762.0, + "completions/mean_length": 516.875, + "completions/min_length": 236.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 387.3333435058594, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 934.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3976736068725586, + "rewards/reward_func/std": 0.2504737079143524, + "reward": 0.3976736068725586, + "reward_std": 0.25047367811203003, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0032803795766085386, + "sampling/sampling_logp_difference/max": 0.46134090423583984, + "sampling/importance_sampling_ratio/min": 0.40576884150505066, + "sampling/importance_sampling_ratio/mean": 0.7740308046340942, + "sampling/importance_sampling_ratio/max": 1.3963788747787476, + "kl": 0.011824634395452449, + "entropy": 0.07033092447090894, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.528179533779621, + "epoch": 0.00412109375, + "step": 211 + }, + { + "loss": 0.4068470597267151, + "grad_norm": 3.2350351810455322, + "learning_rate": 4.846153846153846e-07, + "num_tokens": 2105539.0, + "completions/mean_length": 411.875, + "completions/min_length": 211.0, + "completions/max_length": 894.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 411.875, + "completions/min_terminated_length": 211.0, + "completions/max_terminated_length": 894.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6336805820465088, + "rewards/reward_func/std": 0.2104911506175995, + "reward": 0.6336805820465088, + "reward_std": 0.2104911357164383, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004245626740157604, + "sampling/sampling_logp_difference/max": 0.536460280418396, + "sampling/importance_sampling_ratio/min": 0.4217967391014099, + "sampling/importance_sampling_ratio/mean": 0.8470560312271118, + "sampling/importance_sampling_ratio/max": 1.314088225364685, + "kl": 0.01189991837600246, + "entropy": 0.07382561638951302, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.185008184984326, + "epoch": 0.004140625, + "step": 212 + }, + { + "loss": 0.13338126242160797, + "grad_norm": 2.9553050994873047, + "learning_rate": 4.82051282051282e-07, + "num_tokens": 2115219.0, + "completions/mean_length": 524.125, + "completions/min_length": 239.0, + "completions/max_length": 937.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 391.16668701171875, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 903.0, + "tools/call_frequency": 12.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4602273106575012, + "rewards/reward_func/std": 0.3056102991104126, + "reward": 0.4602273106575012, + "reward_std": 0.305610328912735, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00318622007034719, + "sampling/sampling_logp_difference/max": 0.48520374298095703, + "sampling/importance_sampling_ratio/min": 0.35298991203308105, + "sampling/importance_sampling_ratio/mean": 0.777704119682312, + "sampling/importance_sampling_ratio/max": 1.3875147104263306, + "kl": 0.008244184486102313, + "entropy": 0.06291573744965717, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.057067144662142, + "epoch": 0.00416015625, + "step": 213 + }, + { + "loss": 0.29899275302886963, + "grad_norm": 2.0950863361358643, + "learning_rate": 4.794871794871795e-07, + "num_tokens": 2125450.0, + "completions/mean_length": 593.125, + "completions/min_length": 225.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 263.25, + "completions/min_terminated_length": 225.0, + "completions/max_terminated_length": 340.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4156249761581421, + "rewards/reward_func/std": 0.40199002623558044, + "reward": 0.4156249761581421, + "reward_std": 0.40199002623558044, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002616757759824395, + "sampling/sampling_logp_difference/max": 0.7464618682861328, + "sampling/importance_sampling_ratio/min": 0.7044081687927246, + "sampling/importance_sampling_ratio/mean": 1.073548436164856, + "sampling/importance_sampling_ratio/max": 1.348494052886963, + "kl": 0.006521873947349377, + "entropy": 0.044454383256379515, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.473309975117445, + "epoch": 0.0041796875, + "step": 214 + }, + { + "loss": -0.06223463639616966, + "grad_norm": 2.699631452560425, + "learning_rate": 4.769230769230769e-07, + "num_tokens": 2134887.0, + "completions/mean_length": 494.25, + "completions/min_length": 179.0, + "completions/max_length": 980.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 231.1999969482422, + "completions/min_terminated_length": 179.0, + "completions/max_terminated_length": 268.0, + "tools/call_frequency": 11.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33510732650756836, + "rewards/reward_func/std": 0.2929497957229614, + "reward": 0.33510732650756836, + "reward_std": 0.2929497957229614, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028854392003268003, + "sampling/sampling_logp_difference/max": 0.6863237619400024, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.7319954037666321, + "sampling/importance_sampling_ratio/max": 1.4499869346618652, + "kl": 0.020986638803151436, + "entropy": 0.07259920769138262, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.608661314472556, + "epoch": 0.00419921875, + "step": 215 + }, + { + "loss": 0.4374210834503174, + "grad_norm": 9.106273651123047, + "learning_rate": 4.743589743589743e-07, + "num_tokens": 2144947.0, + "completions/mean_length": 572.875, + "completions/min_length": 232.0, + "completions/max_length": 914.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 377.6000061035156, + "completions/min_terminated_length": 232.0, + "completions/max_terminated_length": 892.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47791171073913574, + "rewards/reward_func/std": 0.3667765259742737, + "reward": 0.47791171073913574, + "reward_std": 0.3667765259742737, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0027058247942477465, + "sampling/sampling_logp_difference/max": 2.005730390548706, + "sampling/importance_sampling_ratio/min": 0.17137998342514038, + "sampling/importance_sampling_ratio/mean": 0.8532277345657349, + "sampling/importance_sampling_ratio/max": 1.6487269401550293, + "kl": 0.00999430259980727, + "entropy": 0.05350906716194004, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.506207996979356, + "epoch": 0.00421875, + "step": 216 + }, + { + "loss": 0.19728338718414307, + "grad_norm": 2.1226253509521484, + "learning_rate": 4.7179487179487176e-07, + "num_tokens": 2153770.0, + "completions/mean_length": 418.375, + "completions/min_length": 181.0, + "completions/max_length": 916.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 255.33334350585938, + "completions/min_terminated_length": 181.0, + "completions/max_terminated_length": 313.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5343005657196045, + "rewards/reward_func/std": 0.3115735352039337, + "reward": 0.5343005657196045, + "reward_std": 0.3115735352039337, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005476216319948435, + "sampling/sampling_logp_difference/max": 0.6627951860427856, + "sampling/importance_sampling_ratio/min": 0.2253536581993103, + "sampling/importance_sampling_ratio/mean": 0.636397659778595, + "sampling/importance_sampling_ratio/max": 1.1224393844604492, + "kl": 0.01013244187925011, + "entropy": 0.09011743264272809, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.842706799507141, + "epoch": 0.00423828125, + "step": 217 + }, + { + "loss": 0.3027743399143219, + "grad_norm": 2.64968204498291, + "learning_rate": 4.692307692307692e-07, + "num_tokens": 2163450.0, + "completions/mean_length": 525.25, + "completions/min_length": 198.0, + "completions/max_length": 979.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 275.20001220703125, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 352.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.26944446563720703, + "rewards/reward_func/std": 0.2802538573741913, + "reward": 0.26944446563720703, + "reward_std": 0.2802538573741913, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034756194800138474, + "sampling/sampling_logp_difference/max": 0.4528195858001709, + "sampling/importance_sampling_ratio/min": 0.2546273171901703, + "sampling/importance_sampling_ratio/mean": 0.9263761043548584, + "sampling/importance_sampling_ratio/max": 1.6455309391021729, + "kl": 0.013089905434753746, + "entropy": 0.06231517472770065, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.118939399719238, + "epoch": 0.0042578125, + "step": 218 + }, + { + "loss": 0.22266344726085663, + "grad_norm": 2.8171796798706055, + "learning_rate": 4.6666666666666666e-07, + "num_tokens": 2173563.0, + "completions/mean_length": 579.125, + "completions/min_length": 206.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 463.5, + "completions/min_terminated_length": 206.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5494791865348816, + "rewards/reward_func/std": 0.32944488525390625, + "reward": 0.5494791865348816, + "reward_std": 0.32944488525390625, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021731869783252478, + "sampling/sampling_logp_difference/max": 0.4366194009780884, + "sampling/importance_sampling_ratio/min": 0.4048827290534973, + "sampling/importance_sampling_ratio/mean": 1.0659887790679932, + "sampling/importance_sampling_ratio/max": 2.775434732437134, + "kl": 0.009141604852629825, + "entropy": 0.053716184047516435, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.864074004814029, + "epoch": 0.00427734375, + "step": 219 + }, + { + "loss": 0.49608737230300903, + "grad_norm": 3.3390953540802, + "learning_rate": 4.641025641025641e-07, + "num_tokens": 2182371.0, + "completions/mean_length": 416.375, + "completions/min_length": 211.0, + "completions/max_length": 1001.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 416.375, + "completions/min_terminated_length": 211.0, + "completions/max_terminated_length": 1001.0, + "tools/call_frequency": 9.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.580826997756958, + "rewards/reward_func/std": 0.3429030179977417, + "reward": 0.580826997756958, + "reward_std": 0.3429030478000641, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00403432734310627, + "sampling/sampling_logp_difference/max": 0.49300527572631836, + "sampling/importance_sampling_ratio/min": 0.3230859637260437, + "sampling/importance_sampling_ratio/mean": 0.9783549308776855, + "sampling/importance_sampling_ratio/max": 1.9732611179351807, + "kl": 0.014006767451064661, + "entropy": 0.061946575064212084, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.339816918596625, + "epoch": 0.004296875, + "step": 220 + }, + { + "loss": 0.13323475420475006, + "grad_norm": 1.9215022325515747, + "learning_rate": 4.6153846153846156e-07, + "num_tokens": 2191931.0, + "completions/mean_length": 510.625, + "completions/min_length": 183.0, + "completions/max_length": 947.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 372.8333435058594, + "completions/min_terminated_length": 183.0, + "completions/max_terminated_length": 914.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.40312498807907104, + "rewards/reward_func/std": 0.24692302942276, + "reward": 0.40312498807907104, + "reward_std": 0.2469230443239212, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035880301147699356, + "sampling/sampling_logp_difference/max": 0.44606471061706543, + "sampling/importance_sampling_ratio/min": 0.27958235144615173, + "sampling/importance_sampling_ratio/mean": 1.1953340768814087, + "sampling/importance_sampling_ratio/max": 2.1457667350769043, + "kl": 0.007615134498337284, + "entropy": 0.06319327442906797, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.665012046694756, + "epoch": 0.00431640625, + "step": 221 + }, + { + "loss": 0.36433061957359314, + "grad_norm": 4.7848711013793945, + "learning_rate": 4.5897435897435896e-07, + "num_tokens": 2200921.0, + "completions/mean_length": 437.625, + "completions/min_length": 208.0, + "completions/max_length": 946.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 374.2857360839844, + "completions/min_terminated_length": 208.0, + "completions/max_terminated_length": 946.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.604687511920929, + "rewards/reward_func/std": 0.28932255506515503, + "reward": 0.604687511920929, + "reward_std": 0.28932252526283264, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0049806092865765095, + "sampling/sampling_logp_difference/max": 0.9259911179542542, + "sampling/importance_sampling_ratio/min": 0.23013423383235931, + "sampling/importance_sampling_ratio/mean": 0.8981885313987732, + "sampling/importance_sampling_ratio/max": 2.3765652179718018, + "kl": 0.011283612227998674, + "entropy": 0.08021793677471578, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.663061188533902, + "epoch": 0.0043359375, + "step": 222 + }, + { + "loss": 0.48037606477737427, + "grad_norm": 4.594852447509766, + "learning_rate": 4.5641025641025636e-07, + "num_tokens": 2209787.0, + "completions/mean_length": 423.5, + "completions/min_length": 183.0, + "completions/max_length": 916.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 264.66668701171875, + "completions/min_terminated_length": 183.0, + "completions/max_terminated_length": 308.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6437499523162842, + "rewards/reward_func/std": 0.2851308584213257, + "reward": 0.6437499523162842, + "reward_std": 0.2851308584213257, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004855136852711439, + "sampling/sampling_logp_difference/max": 0.9259939789772034, + "sampling/importance_sampling_ratio/min": 0.22669252753257751, + "sampling/importance_sampling_ratio/mean": 1.1106401681900024, + "sampling/importance_sampling_ratio/max": 2.539088487625122, + "kl": 0.009186464827507734, + "entropy": 0.10269802145194262, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.527162317186594, + "epoch": 0.00435546875, + "step": 223 + }, + { + "loss": -0.030914306640625, + "grad_norm": 2.377105712890625, + "learning_rate": 4.538461538461538e-07, + "num_tokens": 2220453.0, + "completions/mean_length": 647.375, + "completions/min_length": 221.0, + "completions/max_length": 918.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 395.5, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 857.0, + "tools/call_frequency": 15.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.514484167098999, + "rewards/reward_func/std": 0.3333475887775421, + "reward": 0.514484167098999, + "reward_std": 0.3333475887775421, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025513065047562122, + "sampling/sampling_logp_difference/max": 1.6101272106170654, + "sampling/importance_sampling_ratio/min": 0.11873200535774231, + "sampling/importance_sampling_ratio/mean": 1.0498480796813965, + "sampling/importance_sampling_ratio/max": 2.7985599040985107, + "kl": 0.005778043661848642, + "entropy": 0.05139501683879644, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.189558735117316, + "epoch": 0.004375, + "step": 224 + }, + { + "loss": 0.255688339471817, + "grad_norm": 2.1821401119232178, + "learning_rate": 4.5128205128205125e-07, + "num_tokens": 2229552.0, + "completions/mean_length": 452.5, + "completions/min_length": 224.0, + "completions/max_length": 967.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 384.5714416503906, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 967.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5041666626930237, + "rewards/reward_func/std": 0.3475538194179535, + "reward": 0.5041666626930237, + "reward_std": 0.3475537896156311, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004488230217248201, + "sampling/sampling_logp_difference/max": 0.6478831768035889, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.5875450372695923, + "sampling/importance_sampling_ratio/max": 0.9698420763015747, + "kl": 0.009806208472582512, + "entropy": 0.07768145576119423, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.676495138555765, + "epoch": 0.00439453125, + "step": 225 + }, + { + "loss": 0.18193049728870392, + "grad_norm": 1.1092195510864258, + "learning_rate": 4.487179487179487e-07, + "num_tokens": 2240910.0, + "completions/mean_length": 735.625, + "completions/min_length": 233.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 242.0, + "completions/min_terminated_length": 233.0, + "completions/max_terminated_length": 251.0, + "tools/call_frequency": 16.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5234375, + "rewards/reward_func/std": 0.3577510416507721, + "reward": 0.5234375, + "reward_std": 0.3577510118484497, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001276204944588244, + "sampling/sampling_logp_difference/max": 0.32627129554748535, + "sampling/importance_sampling_ratio/min": 0.39823588728904724, + "sampling/importance_sampling_ratio/mean": 0.758627712726593, + "sampling/importance_sampling_ratio/max": 1.351973295211792, + "kl": 0.004181739834166365, + "entropy": 0.03533317783148959, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.359596062451601, + "epoch": 0.0044140625, + "step": 226 + }, + { + "loss": 0.07422790676355362, + "grad_norm": 1.1971969604492188, + "learning_rate": 4.4615384615384615e-07, + "num_tokens": 2251412.0, + "completions/mean_length": 626.625, + "completions/min_length": 302.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 328.5, + "completions/min_terminated_length": 302.0, + "completions/max_terminated_length": 377.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4443497657775879, + "rewards/reward_func/std": 0.22174546122550964, + "reward": 0.4443497657775879, + "reward_std": 0.22174546122550964, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026229533832520247, + "sampling/sampling_logp_difference/max": 0.43997955322265625, + "sampling/importance_sampling_ratio/min": 0.3645167052745819, + "sampling/importance_sampling_ratio/mean": 0.9489514827728271, + "sampling/importance_sampling_ratio/max": 1.6282668113708496, + "kl": 0.006774512439733371, + "entropy": 0.054552121902816, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.786547780036926, + "epoch": 0.00443359375, + "step": 227 + }, + { + "loss": 0.14711710810661316, + "grad_norm": 2.051323175430298, + "learning_rate": 4.4358974358974355e-07, + "num_tokens": 2261489.0, + "completions/mean_length": 573.875, + "completions/min_length": 215.0, + "completions/max_length": 901.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 530.4285888671875, + "completions/min_terminated_length": 215.0, + "completions/max_terminated_length": 901.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5814236402511597, + "rewards/reward_func/std": 0.08442387729883194, + "reward": 0.5814236402511597, + "reward_std": 0.08442388474941254, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019133614841848612, + "sampling/sampling_logp_difference/max": 0.34751391410827637, + "sampling/importance_sampling_ratio/min": 0.6454712748527527, + "sampling/importance_sampling_ratio/mean": 0.8795366287231445, + "sampling/importance_sampling_ratio/max": 1.2411437034606934, + "kl": 0.005724710470531136, + "entropy": 0.054176112229470164, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.407669456675649, + "epoch": 0.004453125, + "step": 228 + }, + { + "loss": 0.09572426974773407, + "grad_norm": 1.2372654676437378, + "learning_rate": 4.41025641025641e-07, + "num_tokens": 2273511.0, + "completions/mean_length": 817.0, + "completions/min_length": 208.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.75, + "completions/mean_terminated_length": 558.0, + "completions/min_terminated_length": 208.0, + "completions/max_terminated_length": 908.0, + "tools/call_frequency": 18.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6223958730697632, + "rewards/reward_func/std": 0.305561363697052, + "reward": 0.6223958730697632, + "reward_std": 0.305561363697052, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0011241703759878874, + "sampling/sampling_logp_difference/max": 0.48419189453125, + "sampling/importance_sampling_ratio/min": 0.4705720543861389, + "sampling/importance_sampling_ratio/mean": 0.8843491077423096, + "sampling/importance_sampling_ratio/max": 1.2406880855560303, + "kl": 0.007429784687701613, + "entropy": 0.02161985688144341, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.372501960024238, + "epoch": 0.00447265625, + "step": 229 + }, + { + "loss": 0.07111898064613342, + "grad_norm": 3.9417803287506104, + "learning_rate": 4.3846153846153845e-07, + "num_tokens": 2282786.0, + "completions/mean_length": 473.75, + "completions/min_length": 256.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 333.8333435058594, + "completions/min_terminated_length": 256.0, + "completions/max_terminated_length": 450.0, + "tools/call_frequency": 10.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35468751192092896, + "rewards/reward_func/std": 0.24764202535152435, + "reward": 0.35468751192092896, + "reward_std": 0.24764202535152435, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004826758056879044, + "sampling/sampling_logp_difference/max": 0.5205515623092651, + "sampling/importance_sampling_ratio/min": 0.3092087209224701, + "sampling/importance_sampling_ratio/mean": 0.9416130781173706, + "sampling/importance_sampling_ratio/max": 2.31144642829895, + "kl": 0.005723826696339529, + "entropy": 0.07728139567188919, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.370909685268998, + "epoch": 0.0044921875, + "step": 230 + }, + { + "loss": 0.1171821653842926, + "grad_norm": 3.0929393768310547, + "learning_rate": 4.358974358974359e-07, + "num_tokens": 2291763.0, + "completions/mean_length": 437.625, + "completions/min_length": 241.0, + "completions/max_length": 939.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 276.16668701171875, + "completions/min_terminated_length": 241.0, + "completions/max_terminated_length": 360.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4869791567325592, + "rewards/reward_func/std": 0.3078843057155609, + "reward": 0.4869791567325592, + "reward_std": 0.3078843057155609, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004913581535220146, + "sampling/sampling_logp_difference/max": 0.527815580368042, + "sampling/importance_sampling_ratio/min": 0.3196271061897278, + "sampling/importance_sampling_ratio/mean": 1.196288824081421, + "sampling/importance_sampling_ratio/max": 2.220571994781494, + "kl": 0.013374512433074415, + "entropy": 0.08296344720292836, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.84661110304296, + "epoch": 0.00451171875, + "step": 231 + }, + { + "loss": -0.07512101531028748, + "grad_norm": 1.4189815521240234, + "learning_rate": 4.3333333333333335e-07, + "num_tokens": 2302588.0, + "completions/mean_length": 668.125, + "completions/min_length": 223.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 418.25, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 896.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.46116071939468384, + "rewards/reward_func/std": 0.2992607355117798, + "reward": 0.46116071939468384, + "reward_std": 0.2992607355117798, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017534632934257388, + "sampling/sampling_logp_difference/max": 0.48816490173339844, + "sampling/importance_sampling_ratio/min": 0.6904888153076172, + "sampling/importance_sampling_ratio/mean": 1.21932053565979, + "sampling/importance_sampling_ratio/max": 2.0881500244140625, + "kl": 0.004916931902698707, + "entropy": 0.03643272042972967, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.136628285050392, + "epoch": 0.00453125, + "step": 232 + }, + { + "loss": 0.35481542348861694, + "grad_norm": 1.1769698858261108, + "learning_rate": 4.307692307692308e-07, + "num_tokens": 2313442.0, + "completions/mean_length": 671.625, + "completions/min_length": 199.0, + "completions/max_length": 956.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 261.3333435058594, + "completions/min_terminated_length": 199.0, + "completions/max_terminated_length": 328.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3151041865348816, + "rewards/reward_func/std": 0.2933111786842346, + "reward": 0.3151041865348816, + "reward_std": 0.29331114888191223, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021138424053788185, + "sampling/sampling_logp_difference/max": 0.527845025062561, + "sampling/importance_sampling_ratio/min": 0.41379570960998535, + "sampling/importance_sampling_ratio/mean": 0.8555594682693481, + "sampling/importance_sampling_ratio/max": 1.8266681432724, + "kl": 0.006338623890769668, + "entropy": 0.055898294667713344, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.816147517412901, + "epoch": 0.00455078125, + "step": 233 + }, + { + "loss": 0.512356162071228, + "grad_norm": 2.395411968231201, + "learning_rate": 4.2820512820512814e-07, + "num_tokens": 2322942.0, + "completions/mean_length": 502.375, + "completions/min_length": 142.0, + "completions/max_length": 979.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 356.8333435058594, + "completions/min_terminated_length": 142.0, + "completions/max_terminated_length": 876.0, + "tools/call_frequency": 11.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3999842405319214, + "rewards/reward_func/std": 0.3723122179508209, + "reward": 0.3999842405319214, + "reward_std": 0.3723122179508209, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003319086041301489, + "sampling/sampling_logp_difference/max": 0.7246594429016113, + "sampling/importance_sampling_ratio/min": 0.4823830723762512, + "sampling/importance_sampling_ratio/mean": 1.0704108476638794, + "sampling/importance_sampling_ratio/max": 2.0004148483276367, + "kl": 0.008883146940206643, + "entropy": 0.06458034948445857, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.09026606194675, + "epoch": 0.0045703125, + "step": 234 + }, + { + "loss": 0.1576438546180725, + "grad_norm": 2.9662327766418457, + "learning_rate": 4.256410256410256e-07, + "num_tokens": 2332053.0, + "completions/mean_length": 453.625, + "completions/min_length": 193.0, + "completions/max_length": 978.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 292.3333435058594, + "completions/min_terminated_length": 193.0, + "completions/max_terminated_length": 418.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5301339626312256, + "rewards/reward_func/std": 0.30092400312423706, + "reward": 0.5301339626312256, + "reward_std": 0.3009239733219147, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0039410460740327835, + "sampling/sampling_logp_difference/max": 0.5503783226013184, + "sampling/importance_sampling_ratio/min": 0.29877468943595886, + "sampling/importance_sampling_ratio/mean": 0.8696264624595642, + "sampling/importance_sampling_ratio/max": 1.4733471870422363, + "kl": 0.007757270155707374, + "entropy": 0.07961476838681847, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.030109642073512, + "epoch": 0.00458984375, + "step": 235 + }, + { + "loss": 0.004565984010696411, + "grad_norm": 1.8729052543640137, + "learning_rate": 4.2307692307692304e-07, + "num_tokens": 2340885.0, + "completions/mean_length": 420.25, + "completions/min_length": 223.0, + "completions/max_length": 957.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 420.25, + "completions/min_terminated_length": 223.0, + "completions/max_terminated_length": 957.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5691319704055786, + "rewards/reward_func/std": 0.2283010631799698, + "reward": 0.5691319704055786, + "reward_std": 0.2283010482788086, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004367702640593052, + "sampling/sampling_logp_difference/max": 1.1383628845214844, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.7407803535461426, + "sampling/importance_sampling_ratio/max": 1.3800054788589478, + "kl": 0.008348602306796238, + "entropy": 0.08511903980979696, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.329119535163045, + "epoch": 0.004609375, + "step": 236 + }, + { + "loss": 0.4048570990562439, + "grad_norm": 4.881311893463135, + "learning_rate": 4.205128205128205e-07, + "num_tokens": 2349928.0, + "completions/mean_length": 443.25, + "completions/min_length": 246.0, + "completions/max_length": 915.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 288.16668701171875, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 367.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5321158170700073, + "rewards/reward_func/std": 0.2339739203453064, + "reward": 0.5321158170700073, + "reward_std": 0.23397387564182281, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004582147113978863, + "sampling/sampling_logp_difference/max": 0.47611427307128906, + "sampling/importance_sampling_ratio/min": 0.36691203713417053, + "sampling/importance_sampling_ratio/mean": 0.9484845995903015, + "sampling/importance_sampling_ratio/max": 1.3810784816741943, + "kl": 0.009521654166746885, + "entropy": 0.08780399232637137, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.414915386587381, + "epoch": 0.00462890625, + "step": 237 + }, + { + "loss": -0.10161256790161133, + "grad_norm": 3.1614105701446533, + "learning_rate": 4.1794871794871794e-07, + "num_tokens": 2360736.0, + "completions/mean_length": 666.375, + "completions/min_length": 234.0, + "completions/max_length": 944.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 527.4000244140625, + "completions/min_terminated_length": 234.0, + "completions/max_terminated_length": 925.0, + "tools/call_frequency": 16.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.45017361640930176, + "rewards/reward_func/std": 0.3224773705005646, + "reward": 0.45017361640930176, + "reward_std": 0.3224773406982422, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020652359817177057, + "sampling/sampling_logp_difference/max": 0.5930681228637695, + "sampling/importance_sampling_ratio/min": 0.27699702978134155, + "sampling/importance_sampling_ratio/mean": 0.7446950674057007, + "sampling/importance_sampling_ratio/max": 1.1182587146759033, + "kl": 0.007192874123575166, + "entropy": 0.04069553769659251, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.11507973074913, + "epoch": 0.0046484375, + "step": 238 + }, + { + "loss": 0.1388828307390213, + "grad_norm": 2.5150067806243896, + "learning_rate": 4.153846153846154e-07, + "num_tokens": 2371659.0, + "completions/mean_length": 679.625, + "completions/min_length": 271.0, + "completions/max_length": 965.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 450.25, + "completions/min_terminated_length": 271.0, + "completions/max_terminated_length": 847.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.48906251788139343, + "rewards/reward_func/std": 0.251305490732193, + "reward": 0.48906251788139343, + "reward_std": 0.251305490732193, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00321560213342309, + "sampling/sampling_logp_difference/max": 0.5278520584106445, + "sampling/importance_sampling_ratio/min": 0.271382600069046, + "sampling/importance_sampling_ratio/mean": 0.7424805164337158, + "sampling/importance_sampling_ratio/max": 1.1371623277664185, + "kl": 0.008122878120047972, + "entropy": 0.059195323672611266, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.907634576782584, + "epoch": 0.00466796875, + "step": 239 + }, + { + "loss": -0.4788946211338043, + "grad_norm": 6.275322437286377, + "learning_rate": 4.128205128205128e-07, + "num_tokens": 2380013.0, + "completions/mean_length": 359.25, + "completions/min_length": 212.0, + "completions/max_length": 890.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 359.25, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 890.0, + "tools/call_frequency": 9.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6002272367477417, + "rewards/reward_func/std": 0.188318133354187, + "reward": 0.6002272367477417, + "reward_std": 0.18831810355186462, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005844885017722845, + "sampling/sampling_logp_difference/max": 1.1105551719665527, + "sampling/importance_sampling_ratio/min": 0.423122376203537, + "sampling/importance_sampling_ratio/mean": 1.2112560272216797, + "sampling/importance_sampling_ratio/max": 2.051976203918457, + "kl": 0.01426713919499889, + "entropy": 0.08779976610094309, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.897212959825993, + "epoch": 0.0046875, + "step": 240 + }, + { + "loss": -0.14477989077568054, + "grad_norm": 1.0663548707962036, + "learning_rate": 4.1025641025641024e-07, + "num_tokens": 2390195.0, + "completions/mean_length": 587.25, + "completions/min_length": 213.0, + "completions/max_length": 898.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 285.25, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 372.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6129306554794312, + "rewards/reward_func/std": 0.2544763386249542, + "reward": 0.6129306554794312, + "reward_std": 0.2544763386249542, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0027642371132969856, + "sampling/sampling_logp_difference/max": 1.310044288635254, + "sampling/importance_sampling_ratio/min": 0.14236924052238464, + "sampling/importance_sampling_ratio/mean": 0.5599613785743713, + "sampling/importance_sampling_ratio/max": 1.1254678964614868, + "kl": 0.006940011197002605, + "entropy": 0.05077404418261722, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.74030363932252, + "epoch": 0.00470703125, + "step": 241 + }, + { + "loss": -0.23775190114974976, + "grad_norm": 3.230459451675415, + "learning_rate": 4.076923076923077e-07, + "num_tokens": 2399547.0, + "completions/mean_length": 482.375, + "completions/min_length": 182.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 241.1999969482422, + "completions/min_terminated_length": 182.0, + "completions/max_terminated_length": 277.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5742311477661133, + "rewards/reward_func/std": 0.2928982973098755, + "reward": 0.5742311477661133, + "reward_std": 0.2928982973098755, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023718485608696938, + "sampling/sampling_logp_difference/max": 0.4630138874053955, + "sampling/importance_sampling_ratio/min": 0.857441782951355, + "sampling/importance_sampling_ratio/mean": 1.2286694049835205, + "sampling/importance_sampling_ratio/max": 1.8941987752914429, + "kl": 0.012796511931810528, + "entropy": 0.05691756302258, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.41270363330841, + "epoch": 0.0047265625, + "step": 242 + }, + { + "loss": 0.3372679054737091, + "grad_norm": 2.34895658493042, + "learning_rate": 4.0512820512820514e-07, + "num_tokens": 2410013.0, + "completions/mean_length": 622.25, + "completions/min_length": 232.0, + "completions/max_length": 949.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 431.6000061035156, + "completions/min_terminated_length": 232.0, + "completions/max_terminated_length": 860.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.34236109256744385, + "rewards/reward_func/std": 0.27599725127220154, + "reward": 0.34236109256744385, + "reward_std": 0.2759972810745239, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00360473501496017, + "sampling/sampling_logp_difference/max": 0.6249983310699463, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.851478099822998, + "sampling/importance_sampling_ratio/max": 1.7798311710357666, + "kl": 0.009348240215331316, + "entropy": 0.05944643181283027, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.039077389985323, + "epoch": 0.00474609375, + "step": 243 + }, + { + "loss": -0.22189359366893768, + "grad_norm": 2.9364852905273438, + "learning_rate": 4.025641025641026e-07, + "num_tokens": 2418226.0, + "completions/mean_length": 341.5, + "completions/min_length": 200.0, + "completions/max_length": 882.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 264.2857360839844, + "completions/min_terminated_length": 200.0, + "completions/max_terminated_length": 329.0, + "tools/call_frequency": 8.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5729166269302368, + "rewards/reward_func/std": 0.18675115704536438, + "reward": 0.5729166269302368, + "reward_std": 0.18675115704536438, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005369723308831453, + "sampling/sampling_logp_difference/max": 0.47071266174316406, + "sampling/importance_sampling_ratio/min": 0.36565402150154114, + "sampling/importance_sampling_ratio/mean": 1.0711894035339355, + "sampling/importance_sampling_ratio/max": 2.6868669986724854, + "kl": 0.01024444232461974, + "entropy": 0.09190249058883637, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.13576545752585, + "epoch": 0.004765625, + "step": 244 + }, + { + "loss": 0.3068428039550781, + "grad_norm": 2.608705997467041, + "learning_rate": 4e-07, + "num_tokens": 2427024.0, + "completions/mean_length": 414.5, + "completions/min_length": 175.0, + "completions/max_length": 895.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 345.8571472167969, + "completions/min_terminated_length": 175.0, + "completions/max_terminated_length": 886.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.48103538155555725, + "rewards/reward_func/std": 0.3122040331363678, + "reward": 0.48103538155555725, + "reward_std": 0.3122040331363678, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034372119698673487, + "sampling/sampling_logp_difference/max": 0.26412951946258545, + "sampling/importance_sampling_ratio/min": 0.5051968097686768, + "sampling/importance_sampling_ratio/mean": 0.849982738494873, + "sampling/importance_sampling_ratio/max": 1.6420848369598389, + "kl": 0.016208237590035424, + "entropy": 0.0827412762446329, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.456030040979385, + "epoch": 0.00478515625, + "step": 245 + }, + { + "loss": 0.5452172756195068, + "grad_norm": 3.2701520919799805, + "learning_rate": 3.974358974358974e-07, + "num_tokens": 2435372.0, + "completions/mean_length": 359.375, + "completions/min_length": 212.0, + "completions/max_length": 951.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 274.8571472167969, + "completions/min_terminated_length": 212.0, + "completions/max_terminated_length": 316.0, + "tools/call_frequency": 8.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4722456634044647, + "rewards/reward_func/std": 0.2897709310054779, + "reward": 0.4722456634044647, + "reward_std": 0.2897709310054779, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005395137704908848, + "sampling/sampling_logp_difference/max": 0.518953800201416, + "sampling/importance_sampling_ratio/min": 0.49498307704925537, + "sampling/importance_sampling_ratio/mean": 1.0545661449432373, + "sampling/importance_sampling_ratio/max": 2.0728230476379395, + "kl": 0.015186717791948467, + "entropy": 0.09129723603837192, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.902400290593505, + "epoch": 0.0048046875, + "step": 246 + }, + { + "loss": -0.4133989214897156, + "grad_norm": 8.700910568237305, + "learning_rate": 3.9487179487179483e-07, + "num_tokens": 2443546.0, + "completions/mean_length": 335.625, + "completions/min_length": 182.0, + "completions/max_length": 891.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 335.625, + "completions/min_terminated_length": 182.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 8.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5667658448219299, + "rewards/reward_func/std": 0.2736150622367859, + "reward": 0.5667658448219299, + "reward_std": 0.2736150622367859, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005509411450475454, + "sampling/sampling_logp_difference/max": 0.8547611236572266, + "sampling/importance_sampling_ratio/min": 0.37568241357803345, + "sampling/importance_sampling_ratio/mean": 1.0594472885131836, + "sampling/importance_sampling_ratio/max": 2.9738452434539795, + "kl": 0.019708842126419768, + "entropy": 0.07792762917233631, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.9821135122329, + "epoch": 0.00482421875, + "step": 247 + }, + { + "loss": 0.24439018964767456, + "grad_norm": 1.7314670085906982, + "learning_rate": 3.923076923076923e-07, + "num_tokens": 2453047.0, + "completions/mean_length": 503.125, + "completions/min_length": 239.0, + "completions/max_length": 939.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 453.14288330078125, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 939.0, + "tools/call_frequency": 12.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.41371527314186096, + "rewards/reward_func/std": 0.2701904773712158, + "reward": 0.41371527314186096, + "reward_std": 0.27019044756889343, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003044240642338991, + "sampling/sampling_logp_difference/max": 0.7091536521911621, + "sampling/importance_sampling_ratio/min": 0.414340615272522, + "sampling/importance_sampling_ratio/mean": 0.8398758172988892, + "sampling/importance_sampling_ratio/max": 1.830376386642456, + "kl": 0.005543721315916628, + "entropy": 0.06743654439924285, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.418243186548352, + "epoch": 0.00484375, + "step": 248 + }, + { + "loss": 0.04582594335079193, + "grad_norm": 1.9488670825958252, + "learning_rate": 3.8974358974358973e-07, + "num_tokens": 2463296.0, + "completions/mean_length": 595.75, + "completions/min_length": 242.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 549.4285888671875, + "completions/min_terminated_length": 242.0, + "completions/max_terminated_length": 953.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47951388359069824, + "rewards/reward_func/std": 0.33935675024986267, + "reward": 0.47951388359069824, + "reward_std": 0.33935675024986267, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024049924686551094, + "sampling/sampling_logp_difference/max": 0.46229124069213867, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6948279142379761, + "sampling/importance_sampling_ratio/max": 1.3016730546951294, + "kl": 0.0106998011469841, + "entropy": 0.051624147628899664, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.694367308169603, + "epoch": 0.00486328125, + "step": 249 + }, + { + "loss": 0.6917842030525208, + "grad_norm": 3.593508243560791, + "learning_rate": 3.871794871794872e-07, + "num_tokens": 2472906.0, + "completions/mean_length": 516.0, + "completions/min_length": 240.0, + "completions/max_length": 977.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 254.0, + "completions/min_terminated_length": 240.0, + "completions/max_terminated_length": 278.0, + "tools/call_frequency": 11.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5460763573646545, + "rewards/reward_func/std": 0.37325894832611084, + "reward": 0.5460763573646545, + "reward_std": 0.37325894832611084, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035421608481556177, + "sampling/sampling_logp_difference/max": 0.6627938747406006, + "sampling/importance_sampling_ratio/min": 0.548694908618927, + "sampling/importance_sampling_ratio/mean": 1.0473204851150513, + "sampling/importance_sampling_ratio/max": 2.28184175491333, + "kl": 0.012046431860653684, + "entropy": 0.0789710528915748, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.390315119177103, + "epoch": 0.0048828125, + "step": 250 + }, + { + "loss": 0.2545554041862488, + "grad_norm": 1.7696690559387207, + "learning_rate": 3.8461538461538463e-07, + "num_tokens": 2483768.0, + "completions/mean_length": 672.25, + "completions/min_length": 245.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 534.0, + "completions/min_terminated_length": 245.0, + "completions/max_terminated_length": 933.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33125001192092896, + "rewards/reward_func/std": 0.32943403720855713, + "reward": 0.33125001192092896, + "reward_std": 0.32943400740623474, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0021418847609311342, + "sampling/sampling_logp_difference/max": 0.9017462730407715, + "sampling/importance_sampling_ratio/min": 0.5466343760490417, + "sampling/importance_sampling_ratio/mean": 0.8645294308662415, + "sampling/importance_sampling_ratio/max": 1.2763631343841553, + "kl": 0.00468914549855981, + "entropy": 0.052007376158144325, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.270513223484159, + "epoch": 0.00490234375, + "step": 251 + }, + { + "loss": 0.15431508421897888, + "grad_norm": 3.451720952987671, + "learning_rate": 3.82051282051282e-07, + "num_tokens": 2492708.0, + "completions/mean_length": 432.875, + "completions/min_length": 207.0, + "completions/max_length": 925.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 274.0, + "completions/min_terminated_length": 207.0, + "completions/max_terminated_length": 321.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6557291746139526, + "rewards/reward_func/std": 0.22950665652751923, + "reward": 0.6557291746139526, + "reward_std": 0.22950665652751923, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003841268364340067, + "sampling/sampling_logp_difference/max": 0.4676189422607422, + "sampling/importance_sampling_ratio/min": 0.3181956112384796, + "sampling/importance_sampling_ratio/mean": 0.7525894641876221, + "sampling/importance_sampling_ratio/max": 1.1588466167449951, + "kl": 0.007321541866986081, + "entropy": 0.06535267014987767, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.701405875384808, + "epoch": 0.004921875, + "step": 252 + }, + { + "loss": 0.9204708337783813, + "grad_norm": 4.1245951652526855, + "learning_rate": 3.7948717948717947e-07, + "num_tokens": 2501954.0, + "completions/mean_length": 470.125, + "completions/min_length": 186.0, + "completions/max_length": 913.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 336.5, + "completions/min_terminated_length": 186.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5260416865348816, + "rewards/reward_func/std": 0.409084290266037, + "reward": 0.5260416865348816, + "reward_std": 0.409084290266037, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002706964034587145, + "sampling/sampling_logp_difference/max": 0.6249992847442627, + "sampling/importance_sampling_ratio/min": 0.5651641488075256, + "sampling/importance_sampling_ratio/mean": 1.3714429140090942, + "sampling/importance_sampling_ratio/max": 2.911283254623413, + "kl": 0.008900911911041476, + "entropy": 0.05562360223848373, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.097462335601449, + "epoch": 0.00494140625, + "step": 253 + }, + { + "loss": 0.11790582537651062, + "grad_norm": 3.1547703742980957, + "learning_rate": 3.769230769230769e-07, + "num_tokens": 2510958.0, + "completions/mean_length": 440.25, + "completions/min_length": 266.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 278.3333435058594, + "completions/min_terminated_length": 266.0, + "completions/max_terminated_length": 299.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5048958659172058, + "rewards/reward_func/std": 0.2971346080303192, + "reward": 0.5048958659172058, + "reward_std": 0.2971345782279968, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004226565361022949, + "sampling/sampling_logp_difference/max": 0.4317197799682617, + "sampling/importance_sampling_ratio/min": 0.49727392196655273, + "sampling/importance_sampling_ratio/mean": 1.1198095083236694, + "sampling/importance_sampling_ratio/max": 2.6480343341827393, + "kl": 0.012250670813955367, + "entropy": 0.08418203931069002, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.551768200471997, + "epoch": 0.0049609375, + "step": 254 + }, + { + "loss": 0.08045116066932678, + "grad_norm": 2.4946038722991943, + "learning_rate": 3.743589743589743e-07, + "num_tokens": 2520432.0, + "completions/mean_length": 499.0, + "completions/min_length": 189.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 442.14288330078125, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6484325528144836, + "rewards/reward_func/std": 0.13834287226200104, + "reward": 0.6484325528144836, + "reward_std": 0.13834288716316223, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003627793863415718, + "sampling/sampling_logp_difference/max": 0.4617626667022705, + "sampling/importance_sampling_ratio/min": 0.2289419025182724, + "sampling/importance_sampling_ratio/mean": 0.8708775639533997, + "sampling/importance_sampling_ratio/max": 1.336158275604248, + "kl": 0.010760653887700755, + "entropy": 0.06451501155970618, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.795375980436802, + "epoch": 0.00498046875, + "step": 255 + }, + { + "loss": 0.3622209429740906, + "grad_norm": 2.6884613037109375, + "learning_rate": 3.7179487179487177e-07, + "num_tokens": 2530751.0, + "completions/mean_length": 603.25, + "completions/min_length": 226.0, + "completions/max_length": 951.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 406.20001220703125, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 945.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.2697916626930237, + "rewards/reward_func/std": 0.26207685470581055, + "reward": 0.2697916626930237, + "reward_std": 0.26207685470581055, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026674973778426647, + "sampling/sampling_logp_difference/max": 0.7404665946960449, + "sampling/importance_sampling_ratio/min": 0.4340439736843109, + "sampling/importance_sampling_ratio/mean": 0.9640856385231018, + "sampling/importance_sampling_ratio/max": 1.9910809993743896, + "kl": 0.008347249669895973, + "entropy": 0.05282614310272038, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.763602536171675, + "epoch": 0.005, + "step": 256 + }, + { + "loss": -0.10057283937931061, + "grad_norm": 1.7955888509750366, + "learning_rate": 3.692307692307692e-07, + "num_tokens": 2540327.0, + "completions/mean_length": 512.25, + "completions/min_length": 219.0, + "completions/max_length": 958.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 457.857177734375, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 958.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5256944298744202, + "rewards/reward_func/std": 0.2242835909128189, + "reward": 0.5256944298744202, + "reward_std": 0.2242835909128189, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030779344961047173, + "sampling/sampling_logp_difference/max": 0.7177464962005615, + "sampling/importance_sampling_ratio/min": 0.316733717918396, + "sampling/importance_sampling_ratio/mean": 1.041670322418213, + "sampling/importance_sampling_ratio/max": 1.9397798776626587, + "kl": 0.006755458176485263, + "entropy": 0.06296728667803109, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.611780937761068, + "epoch": 0.00501953125, + "step": 257 + }, + { + "loss": 0.13883616030216217, + "grad_norm": 2.0978949069976807, + "learning_rate": 3.666666666666666e-07, + "num_tokens": 2552291.0, + "completions/mean_length": 810.625, + "completions/min_length": 253.0, + "completions/max_length": 948.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 702.25, + "completions/min_terminated_length": 253.0, + "completions/max_terminated_length": 926.0, + "tools/call_frequency": 18.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.12708334624767303, + "rewards/reward_func/std": 0.2180245965719223, + "reward": 0.12708334624767303, + "reward_std": 0.2180245965719223, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018076448468491435, + "sampling/sampling_logp_difference/max": 0.48273229598999023, + "sampling/importance_sampling_ratio/min": 0.2916221618652344, + "sampling/importance_sampling_ratio/mean": 0.7896173000335693, + "sampling/importance_sampling_ratio/max": 1.239085078239441, + "kl": 0.004224126358167268, + "entropy": 0.025223220349289477, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.952659163624048, + "epoch": 0.0050390625, + "step": 258 + }, + { + "loss": 0.20409919321537018, + "grad_norm": 5.991527080535889, + "learning_rate": 3.6410256410256406e-07, + "num_tokens": 2562020.0, + "completions/mean_length": 530.75, + "completions/min_length": 236.0, + "completions/max_length": 920.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 406.3333435058594, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 920.0, + "tools/call_frequency": 12.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3999999761581421, + "rewards/reward_func/std": 0.307285875082016, + "reward": 0.3999999761581421, + "reward_std": 0.3072858452796936, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0029359450563788414, + "sampling/sampling_logp_difference/max": 0.4760756492614746, + "sampling/importance_sampling_ratio/min": 0.33844107389450073, + "sampling/importance_sampling_ratio/mean": 0.7692326307296753, + "sampling/importance_sampling_ratio/max": 1.5984328985214233, + "kl": 0.008875918210833333, + "entropy": 0.06782811740413308, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.801711106672883, + "epoch": 0.00505859375, + "step": 259 + }, + { + "loss": 0.6332271099090576, + "grad_norm": 3.890131711959839, + "learning_rate": 3.615384615384615e-07, + "num_tokens": 2570778.0, + "completions/mean_length": 409.125, + "completions/min_length": 161.0, + "completions/max_length": 944.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 409.125, + "completions/min_terminated_length": 161.0, + "completions/max_terminated_length": 944.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.49784228205680847, + "rewards/reward_func/std": 0.355107843875885, + "reward": 0.49784228205680847, + "reward_std": 0.355107843875885, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004275027196854353, + "sampling/sampling_logp_difference/max": 0.3663235902786255, + "sampling/importance_sampling_ratio/min": 0.41311919689178467, + "sampling/importance_sampling_ratio/mean": 0.8995022773742676, + "sampling/importance_sampling_ratio/max": 1.5118647813796997, + "kl": 0.00842065607139375, + "entropy": 0.07047532743308693, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.254022534936666, + "epoch": 0.005078125, + "step": 260 + }, + { + "loss": 0.3946019113063812, + "grad_norm": 1.8134360313415527, + "learning_rate": 3.5897435897435896e-07, + "num_tokens": 2581688.0, + "completions/mean_length": 677.125, + "completions/min_length": 270.0, + "completions/max_length": 931.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 542.7999877929688, + "completions/min_terminated_length": 270.0, + "completions/max_terminated_length": 916.0, + "tools/call_frequency": 16.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.20260417461395264, + "rewards/reward_func/std": 0.21722359955310822, + "reward": 0.20260417461395264, + "reward_std": 0.21722358465194702, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024786368012428284, + "sampling/sampling_logp_difference/max": 0.6597261428833008, + "sampling/importance_sampling_ratio/min": 0.4014762043952942, + "sampling/importance_sampling_ratio/mean": 0.8914144039154053, + "sampling/importance_sampling_ratio/max": 2.060478925704956, + "kl": 0.009480725857429206, + "entropy": 0.04532658477546647, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.665925273671746, + "epoch": 0.00509765625, + "step": 261 + }, + { + "loss": 0.28650856018066406, + "grad_norm": 2.183043956756592, + "learning_rate": 3.564102564102564e-07, + "num_tokens": 2591184.0, + "completions/mean_length": 501.625, + "completions/min_length": 171.0, + "completions/max_length": 910.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 272.6000061035156, + "completions/min_terminated_length": 171.0, + "completions/max_terminated_length": 328.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44479167461395264, + "rewards/reward_func/std": 0.31095051765441895, + "reward": 0.44479167461395264, + "reward_std": 0.31095051765441895, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030949905049055815, + "sampling/sampling_logp_difference/max": 0.36498045921325684, + "sampling/importance_sampling_ratio/min": 0.5009283423423767, + "sampling/importance_sampling_ratio/mean": 0.9438624382019043, + "sampling/importance_sampling_ratio/max": 1.7007322311401367, + "kl": 0.010130164941074327, + "entropy": 0.06667448452208191, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.131530063226819, + "epoch": 0.0051171875, + "step": 262 + }, + { + "loss": 1.2442313432693481, + "grad_norm": 9.383150100708008, + "learning_rate": 3.5384615384615386e-07, + "num_tokens": 2599911.0, + "completions/mean_length": 405.25, + "completions/min_length": 177.0, + "completions/max_length": 917.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 247.0, + "completions/min_terminated_length": 177.0, + "completions/max_terminated_length": 310.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5588542222976685, + "rewards/reward_func/std": 0.24688977003097534, + "reward": 0.5588542222976685, + "reward_std": 0.24688975512981415, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003964157309383154, + "sampling/sampling_logp_difference/max": 0.7733534574508667, + "sampling/importance_sampling_ratio/min": 0.2590709328651428, + "sampling/importance_sampling_ratio/mean": 0.8843687176704407, + "sampling/importance_sampling_ratio/max": 2.191075086593628, + "kl": 0.012059942760970443, + "entropy": 0.06275115488097072, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.320636674761772, + "epoch": 0.00513671875, + "step": 263 + }, + { + "loss": 0.37286993861198425, + "grad_norm": 2.672515869140625, + "learning_rate": 3.5128205128205126e-07, + "num_tokens": 2610135.0, + "completions/mean_length": 593.375, + "completions/min_length": 234.0, + "completions/max_length": 946.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 481.3333435058594, + "completions/min_terminated_length": 234.0, + "completions/max_terminated_length": 920.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3367559611797333, + "rewards/reward_func/std": 0.3070508539676666, + "reward": 0.3367559611797333, + "reward_std": 0.3070508539676666, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024601721670478582, + "sampling/sampling_logp_difference/max": 0.8236117362976074, + "sampling/importance_sampling_ratio/min": 0.19679157435894012, + "sampling/importance_sampling_ratio/mean": 0.5956424474716187, + "sampling/importance_sampling_ratio/max": 1.0321059226989746, + "kl": 0.0078637120386702, + "entropy": 0.05177374114282429, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.141826950013638, + "epoch": 0.00515625, + "step": 264 + }, + { + "loss": 0.14041805267333984, + "grad_norm": 1.8860719203948975, + "learning_rate": 3.487179487179487e-07, + "num_tokens": 2620284.0, + "completions/mean_length": 583.125, + "completions/min_length": 214.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 404.3999938964844, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 929.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.32306545972824097, + "rewards/reward_func/std": 0.2831360101699829, + "reward": 0.32306545972824097, + "reward_std": 0.2831359803676605, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00291975075379014, + "sampling/sampling_logp_difference/max": 0.6686680316925049, + "sampling/importance_sampling_ratio/min": 0.12769438326358795, + "sampling/importance_sampling_ratio/mean": 0.8684192299842834, + "sampling/importance_sampling_ratio/max": 2.537724733352661, + "kl": 0.005063337637693621, + "entropy": 0.05701160233002156, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.60495625063777, + "epoch": 0.00517578125, + "step": 265 + }, + { + "loss": 0.28943103551864624, + "grad_norm": 3.3058063983917236, + "learning_rate": 3.461538461538461e-07, + "num_tokens": 2630253.0, + "completions/mean_length": 559.625, + "completions/min_length": 257.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 439.66668701171875, + "completions/min_terminated_length": 257.0, + "completions/max_terminated_length": 930.0, + "tools/call_frequency": 13.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.39698323607444763, + "rewards/reward_func/std": 0.30874374508857727, + "reward": 0.39698323607444763, + "reward_std": 0.30874374508857727, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003055904060602188, + "sampling/sampling_logp_difference/max": 0.49303483963012695, + "sampling/importance_sampling_ratio/min": 0.7067289352416992, + "sampling/importance_sampling_ratio/mean": 1.060021162033081, + "sampling/importance_sampling_ratio/max": 1.7900673151016235, + "kl": 0.007326426362851635, + "entropy": 0.05664075195090845, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.846521265804768, + "epoch": 0.0051953125, + "step": 266 + }, + { + "loss": 0.1927010864019394, + "grad_norm": 2.789919376373291, + "learning_rate": 3.4358974358974356e-07, + "num_tokens": 2640477.0, + "completions/mean_length": 592.875, + "completions/min_length": 241.0, + "completions/max_length": 956.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 389.20001220703125, + "completions/min_terminated_length": 241.0, + "completions/max_terminated_length": 893.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3309027850627899, + "rewards/reward_func/std": 0.32714641094207764, + "reward": 0.3309027850627899, + "reward_std": 0.32714641094207764, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0029302590992301702, + "sampling/sampling_logp_difference/max": 2.2408249378204346, + "sampling/importance_sampling_ratio/min": 0.22138720750808716, + "sampling/importance_sampling_ratio/mean": 0.8111799955368042, + "sampling/importance_sampling_ratio/max": 1.3457266092300415, + "kl": 0.011697827110765502, + "entropy": 0.054949569108430296, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.70317148976028, + "epoch": 0.00521484375, + "step": 267 + }, + { + "loss": 0.4180086851119995, + "grad_norm": 2.9438531398773193, + "learning_rate": 3.41025641025641e-07, + "num_tokens": 2650221.0, + "completions/mean_length": 532.75, + "completions/min_length": 233.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 305.8000183105469, + "completions/min_terminated_length": 233.0, + "completions/max_terminated_length": 350.0, + "tools/call_frequency": 12.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3185606002807617, + "rewards/reward_func/std": 0.25550103187561035, + "reward": 0.3185606002807617, + "reward_std": 0.25550100207328796, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002574396785348654, + "sampling/sampling_logp_difference/max": 0.6835508346557617, + "sampling/importance_sampling_ratio/min": 0.5677266120910645, + "sampling/importance_sampling_ratio/mean": 0.9919342398643494, + "sampling/importance_sampling_ratio/max": 1.26613450050354, + "kl": 0.008501016622176394, + "entropy": 0.0647088442929089, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.950415801256895, + "epoch": 0.005234375, + "step": 268 + }, + { + "loss": 0.42641839385032654, + "grad_norm": 3.0607430934906006, + "learning_rate": 3.3846153846153845e-07, + "num_tokens": 2659701.0, + "completions/mean_length": 500.5, + "completions/min_length": 210.0, + "completions/max_length": 968.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 442.4285888671875, + "completions/min_terminated_length": 210.0, + "completions/max_terminated_length": 968.0, + "tools/call_frequency": 11.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.41086310148239136, + "rewards/reward_func/std": 0.3460092544555664, + "reward": 0.41086310148239136, + "reward_std": 0.3460092544555664, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0027615963481366634, + "sampling/sampling_logp_difference/max": 0.39921092987060547, + "sampling/importance_sampling_ratio/min": 0.18175601959228516, + "sampling/importance_sampling_ratio/mean": 1.042239785194397, + "sampling/importance_sampling_ratio/max": 1.9296469688415527, + "kl": 0.005965907941572368, + "entropy": 0.0633164590690285, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.95558999106288, + "epoch": 0.00525390625, + "step": 269 + }, + { + "loss": 0.7159923315048218, + "grad_norm": 4.068979740142822, + "learning_rate": 3.3589743589743585e-07, + "num_tokens": 2668590.0, + "completions/mean_length": 426.0, + "completions/min_length": 216.0, + "completions/max_length": 926.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 426.0, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 926.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5032332539558411, + "rewards/reward_func/std": 0.32045239210128784, + "reward": 0.5032332539558411, + "reward_std": 0.32045239210128784, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0038039260543882847, + "sampling/sampling_logp_difference/max": 0.3469514846801758, + "sampling/importance_sampling_ratio/min": 0.4850713014602661, + "sampling/importance_sampling_ratio/mean": 1.5479283332824707, + "sampling/importance_sampling_ratio/max": 2.9766035079956055, + "kl": 0.007165697243181057, + "entropy": 0.06942749413428828, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.65005729533732, + "epoch": 0.0052734375, + "step": 270 + }, + { + "loss": 0.08835620433092117, + "grad_norm": 2.3882877826690674, + "learning_rate": 3.333333333333333e-07, + "num_tokens": 2677522.0, + "completions/mean_length": 431.25, + "completions/min_length": 47.0, + "completions/max_length": 945.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 357.8571472167969, + "completions/min_terminated_length": 47.0, + "completions/max_terminated_length": 901.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35798612236976624, + "rewards/reward_func/std": 0.3249582350254059, + "reward": 0.35798612236976624, + "reward_std": 0.3249582350254059, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004091963637620211, + "sampling/sampling_logp_difference/max": 0.49445104598999023, + "sampling/importance_sampling_ratio/min": 0.38362717628479004, + "sampling/importance_sampling_ratio/mean": 0.7345030903816223, + "sampling/importance_sampling_ratio/max": 2.136629343032837, + "kl": 0.013206218922277912, + "entropy": 0.0803545038215816, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.845717526972294, + "epoch": 0.00529296875, + "step": 271 + }, + { + "loss": -0.17715829610824585, + "grad_norm": 8.311458587646484, + "learning_rate": 3.3076923076923075e-07, + "num_tokens": 2685978.0, + "completions/mean_length": 372.125, + "completions/min_length": 248.0, + "completions/max_length": 882.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 299.2857360839844, + "completions/min_terminated_length": 248.0, + "completions/max_terminated_length": 338.0, + "tools/call_frequency": 9.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4957291781902313, + "rewards/reward_func/std": 0.21234259009361267, + "reward": 0.4957291781902313, + "reward_std": 0.21234257519245148, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005504378117620945, + "sampling/sampling_logp_difference/max": 0.388149619102478, + "sampling/importance_sampling_ratio/min": 0.3032427132129669, + "sampling/importance_sampling_ratio/mean": 1.2192370891571045, + "sampling/importance_sampling_ratio/max": 2.9640932083129883, + "kl": 0.009147482982371002, + "entropy": 0.08134911931119859, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.176749652251601, + "epoch": 0.0053125, + "step": 272 + }, + { + "loss": 0.5521776080131531, + "grad_norm": 1.8538135290145874, + "learning_rate": 3.282051282051282e-07, + "num_tokens": 2696202.0, + "completions/mean_length": 592.75, + "completions/min_length": 217.0, + "completions/max_length": 946.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 542.2857666015625, + "completions/min_terminated_length": 217.0, + "completions/max_terminated_length": 933.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3427083492279053, + "rewards/reward_func/std": 0.323713481426239, + "reward": 0.3427083492279053, + "reward_std": 0.3237135112285614, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002635449869558215, + "sampling/sampling_logp_difference/max": 0.5080971717834473, + "sampling/importance_sampling_ratio/min": 0.6071144342422485, + "sampling/importance_sampling_ratio/mean": 0.8621105551719666, + "sampling/importance_sampling_ratio/max": 1.2090110778808594, + "kl": 0.005441776484076399, + "entropy": 0.05101310706231743, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.009425513446331, + "epoch": 0.00533203125, + "step": 273 + }, + { + "loss": 0.3215656876564026, + "grad_norm": 3.1587255001068115, + "learning_rate": 3.2564102564102565e-07, + "num_tokens": 2704809.0, + "completions/mean_length": 390.5, + "completions/min_length": 198.0, + "completions/max_length": 940.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 323.2857360839844, + "completions/min_terminated_length": 198.0, + "completions/max_terminated_length": 940.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6639137268066406, + "rewards/reward_func/std": 0.2915262281894684, + "reward": 0.6639137268066406, + "reward_std": 0.2915262281894684, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002956998534500599, + "sampling/sampling_logp_difference/max": 0.3465104103088379, + "sampling/importance_sampling_ratio/min": 0.6935186982154846, + "sampling/importance_sampling_ratio/mean": 1.1657259464263916, + "sampling/importance_sampling_ratio/max": 1.6814404726028442, + "kl": 0.010600298366625793, + "entropy": 0.06563122407533228, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.960654117166996, + "epoch": 0.0053515625, + "step": 274 + }, + { + "loss": -0.3319062888622284, + "grad_norm": 4.9315032958984375, + "learning_rate": 3.230769230769231e-07, + "num_tokens": 2713767.0, + "completions/mean_length": 434.875, + "completions/min_length": 242.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 369.0000305175781, + "completions/min_terminated_length": 242.0, + "completions/max_terminated_length": 871.0, + "tools/call_frequency": 10.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4644097089767456, + "rewards/reward_func/std": 0.34011125564575195, + "reward": 0.4644097089767456, + "reward_std": 0.34011125564575195, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004538896027952433, + "sampling/sampling_logp_difference/max": 0.4556241035461426, + "sampling/importance_sampling_ratio/min": 0.3139016628265381, + "sampling/importance_sampling_ratio/mean": 1.2104302644729614, + "sampling/importance_sampling_ratio/max": 1.9574246406555176, + "kl": 0.012387637660140172, + "entropy": 0.08453356771497056, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.680624648928642, + "epoch": 0.00537109375, + "step": 275 + }, + { + "loss": 0.17267557978630066, + "grad_norm": 3.8823466300964355, + "learning_rate": 3.2051282051282055e-07, + "num_tokens": 2722540.0, + "completions/mean_length": 411.75, + "completions/min_length": 192.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 334.8571472167969, + "completions/min_terminated_length": 192.0, + "completions/max_terminated_length": 911.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.32492560148239136, + "rewards/reward_func/std": 0.4303930103778839, + "reward": 0.32492560148239136, + "reward_std": 0.4303930103778839, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037658889777958393, + "sampling/sampling_logp_difference/max": 0.4811692237854004, + "sampling/importance_sampling_ratio/min": 0.5418630242347717, + "sampling/importance_sampling_ratio/mean": 0.8525827527046204, + "sampling/importance_sampling_ratio/max": 1.4181655645370483, + "kl": 0.014727898655110039, + "entropy": 0.07351790741086006, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.200543215498328, + "epoch": 0.005390625, + "step": 276 + }, + { + "loss": 0.006452344357967377, + "grad_norm": 2.1546685695648193, + "learning_rate": 3.179487179487179e-07, + "num_tokens": 2732922.0, + "completions/mean_length": 611.875, + "completions/min_length": 205.0, + "completions/max_length": 1030.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 409.3999938964844, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 918.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3124256134033203, + "rewards/reward_func/std": 0.35338807106018066, + "reward": 0.3124256134033203, + "reward_std": 0.35338807106018066, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002445520833134651, + "sampling/sampling_logp_difference/max": 0.6169061660766602, + "sampling/importance_sampling_ratio/min": 0.5349166989326477, + "sampling/importance_sampling_ratio/mean": 0.9325404167175293, + "sampling/importance_sampling_ratio/max": 1.5392853021621704, + "kl": 0.00671390893694479, + "entropy": 0.04692527145380154, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.845683749765158, + "epoch": 0.00541015625, + "step": 277 + }, + { + "loss": 0.2690998911857605, + "grad_norm": 2.1819982528686523, + "learning_rate": 3.1538461538461534e-07, + "num_tokens": 2743233.0, + "completions/mean_length": 603.125, + "completions/min_length": 277.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 426.0, + "completions/min_terminated_length": 277.0, + "completions/max_terminated_length": 906.0, + "tools/call_frequency": 14.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4230903089046478, + "rewards/reward_func/std": 0.29558491706848145, + "reward": 0.4230903089046478, + "reward_std": 0.29558491706848145, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018546689534559846, + "sampling/sampling_logp_difference/max": 0.3528444766998291, + "sampling/importance_sampling_ratio/min": 0.4631556570529938, + "sampling/importance_sampling_ratio/mean": 0.9949965476989746, + "sampling/importance_sampling_ratio/max": 1.3670542240142822, + "kl": 0.006985101485042833, + "entropy": 0.04386795015307143, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.61746733263135, + "epoch": 0.0054296875, + "step": 278 + }, + { + "loss": 0.49024343490600586, + "grad_norm": 2.2857329845428467, + "learning_rate": 3.128205128205128e-07, + "num_tokens": 2752234.0, + "completions/mean_length": 439.875, + "completions/min_length": 218.0, + "completions/max_length": 973.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 271.3333435058594, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 321.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4022321403026581, + "rewards/reward_func/std": 0.3085748553276062, + "reward": 0.4022321403026581, + "reward_std": 0.3085748553276062, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003428212832659483, + "sampling/sampling_logp_difference/max": 0.662794291973114, + "sampling/importance_sampling_ratio/min": 0.5272746086120605, + "sampling/importance_sampling_ratio/mean": 0.9355629682540894, + "sampling/importance_sampling_ratio/max": 1.4208091497421265, + "kl": 0.012711570867395494, + "entropy": 0.06632974615786225, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.280066946521401, + "epoch": 0.00544921875, + "step": 279 + }, + { + "loss": 0.3574988543987274, + "grad_norm": 1.7764419317245483, + "learning_rate": 3.1025641025641024e-07, + "num_tokens": 2762453.0, + "completions/mean_length": 593.5, + "completions/min_length": 238.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 401.0, + "completions/min_terminated_length": 238.0, + "completions/max_terminated_length": 869.0, + "tools/call_frequency": 14.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3966856002807617, + "rewards/reward_func/std": 0.393940806388855, + "reward": 0.3966856002807617, + "reward_std": 0.393940806388855, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002747135702520609, + "sampling/sampling_logp_difference/max": 0.5519626140594482, + "sampling/importance_sampling_ratio/min": 0.2907535135746002, + "sampling/importance_sampling_ratio/mean": 0.8962721228599548, + "sampling/importance_sampling_ratio/max": 1.6953132152557373, + "kl": 0.006852292835901608, + "entropy": 0.05655867321183905, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.556974707171321, + "epoch": 0.00546875, + "step": 280 + }, + { + "loss": 0.7859470248222351, + "grad_norm": 2.682985305786133, + "learning_rate": 3.076923076923077e-07, + "num_tokens": 2772062.0, + "completions/mean_length": 516.875, + "completions/min_length": 209.0, + "completions/max_length": 991.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 456.4285888671875, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 991.0, + "tools/call_frequency": 11.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4873958230018616, + "rewards/reward_func/std": 0.369404137134552, + "reward": 0.4873958230018616, + "reward_std": 0.3694041073322296, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002848046598955989, + "sampling/sampling_logp_difference/max": 0.7733539342880249, + "sampling/importance_sampling_ratio/min": 0.5799451470375061, + "sampling/importance_sampling_ratio/mean": 1.0076072216033936, + "sampling/importance_sampling_ratio/max": 1.6213302612304688, + "kl": 0.009964726748876274, + "entropy": 0.06355726870242506, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.16184863448143, + "epoch": 0.00548828125, + "step": 281 + }, + { + "loss": 0.2257322520017624, + "grad_norm": 1.7871273756027222, + "learning_rate": 3.0512820512820514e-07, + "num_tokens": 2781624.0, + "completions/mean_length": 510.25, + "completions/min_length": 233.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 374.0, + "completions/min_terminated_length": 233.0, + "completions/max_terminated_length": 914.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4098363220691681, + "rewards/reward_func/std": 0.2653016149997711, + "reward": 0.4098363220691681, + "reward_std": 0.26530158519744873, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035622466821223497, + "sampling/sampling_logp_difference/max": 0.48767638206481934, + "sampling/importance_sampling_ratio/min": 0.20117820799350739, + "sampling/importance_sampling_ratio/mean": 0.7318695783615112, + "sampling/importance_sampling_ratio/max": 1.379134178161621, + "kl": 0.011119705435703509, + "entropy": 0.07371784775750712, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.521866815164685, + "epoch": 0.0055078125, + "step": 282 + }, + { + "loss": 0.37720081210136414, + "grad_norm": 2.757375478744507, + "learning_rate": 3.0256410256410254e-07, + "num_tokens": 2791151.0, + "completions/mean_length": 505.875, + "completions/min_length": 226.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 363.8333435058594, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5060268044471741, + "rewards/reward_func/std": 0.283505916595459, + "reward": 0.5060268044471741, + "reward_std": 0.283505916595459, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003229086520150304, + "sampling/sampling_logp_difference/max": 0.9243378639221191, + "sampling/importance_sampling_ratio/min": 0.27915725111961365, + "sampling/importance_sampling_ratio/mean": 1.1513290405273438, + "sampling/importance_sampling_ratio/max": 2.4535470008850098, + "kl": 0.007493420067476109, + "entropy": 0.05738434800878167, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.770787796005607, + "epoch": 0.00552734375, + "step": 283 + }, + { + "loss": 0.03125610947608948, + "grad_norm": 1.7123348712921143, + "learning_rate": 3e-07, + "num_tokens": 2800637.0, + "completions/mean_length": 499.25, + "completions/min_length": 79.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 360.0, + "completions/min_terminated_length": 79.0, + "completions/max_terminated_length": 929.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3230902850627899, + "rewards/reward_func/std": 0.3223940134048462, + "reward": 0.3230902850627899, + "reward_std": 0.3223940134048462, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003757805097848177, + "sampling/sampling_logp_difference/max": 0.7022181749343872, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 1.021155595779419, + "sampling/importance_sampling_ratio/max": 1.6663577556610107, + "kl": 0.009885039769869763, + "entropy": 0.05905633035581559, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.598806582391262, + "epoch": 0.005546875, + "step": 284 + }, + { + "loss": 0.07349050045013428, + "grad_norm": 1.6907917261123657, + "learning_rate": 2.9743589743589744e-07, + "num_tokens": 2810463.0, + "completions/mean_length": 542.0, + "completions/min_length": 284.0, + "completions/max_length": 962.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 493.857177734375, + "completions/min_terminated_length": 284.0, + "completions/max_terminated_length": 962.0, + "tools/call_frequency": 12.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4739583134651184, + "rewards/reward_func/std": 0.2312057614326477, + "reward": 0.4739583134651184, + "reward_std": 0.2312057465314865, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004063331987708807, + "sampling/sampling_logp_difference/max": 0.4973626136779785, + "sampling/importance_sampling_ratio/min": 0.6010317206382751, + "sampling/importance_sampling_ratio/mean": 0.9641481637954712, + "sampling/importance_sampling_ratio/max": 1.8335744142532349, + "kl": 0.009536586163449101, + "entropy": 0.07695904851425439, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.222998347133398, + "epoch": 0.00556640625, + "step": 285 + }, + { + "loss": 0.4153844118118286, + "grad_norm": 2.5573201179504395, + "learning_rate": 2.948717948717949e-07, + "num_tokens": 2820560.0, + "completions/mean_length": 578.375, + "completions/min_length": 214.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 578.375, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 928.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3647569417953491, + "rewards/reward_func/std": 0.35492950677871704, + "reward": 0.3647569417953491, + "reward_std": 0.35492947697639465, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026838169433176517, + "sampling/sampling_logp_difference/max": 0.34679412841796875, + "sampling/importance_sampling_ratio/min": 0.8228551149368286, + "sampling/importance_sampling_ratio/mean": 1.0647764205932617, + "sampling/importance_sampling_ratio/max": 1.5636425018310547, + "kl": 0.011121263742097653, + "entropy": 0.06327253236668184, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.911364421248436, + "epoch": 0.0055859375, + "step": 286 + }, + { + "loss": 0.24922344088554382, + "grad_norm": 2.8553597927093506, + "learning_rate": 2.9230769230769234e-07, + "num_tokens": 2829631.0, + "completions/mean_length": 447.875, + "completions/min_length": 205.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 447.875, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 954.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4235457181930542, + "rewards/reward_func/std": 0.33427393436431885, + "reward": 0.4235457181930542, + "reward_std": 0.33427393436431885, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0046639712527394295, + "sampling/sampling_logp_difference/max": 0.40629053115844727, + "sampling/importance_sampling_ratio/min": 0.5635632872581482, + "sampling/importance_sampling_ratio/mean": 0.8803904056549072, + "sampling/importance_sampling_ratio/max": 1.2746460437774658, + "kl": 0.01030073469155468, + "entropy": 0.09440505714155734, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.013619810342789, + "epoch": 0.00560546875, + "step": 287 + }, + { + "loss": 0.18108324706554413, + "grad_norm": 4.046027183532715, + "learning_rate": 2.8974358974358973e-07, + "num_tokens": 2837994.0, + "completions/mean_length": 360.0, + "completions/min_length": 254.0, + "completions/max_length": 891.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 284.14288330078125, + "completions/min_terminated_length": 254.0, + "completions/max_terminated_length": 309.0, + "tools/call_frequency": 9.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4598958492279053, + "rewards/reward_func/std": 0.2879347801208496, + "reward": 0.4598958492279053, + "reward_std": 0.2879347801208496, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005825299769639969, + "sampling/sampling_logp_difference/max": 0.5609333515167236, + "sampling/importance_sampling_ratio/min": 0.4905644655227661, + "sampling/importance_sampling_ratio/mean": 1.072709321975708, + "sampling/importance_sampling_ratio/max": 1.9503333568572998, + "kl": 0.009712367376778275, + "entropy": 0.08907330129295588, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.494555966928601, + "epoch": 0.005625, + "step": 288 + }, + { + "loss": 0.06854607164859772, + "grad_norm": 2.4128007888793945, + "learning_rate": 2.8717948717948713e-07, + "num_tokens": 2848131.0, + "completions/mean_length": 583.0, + "completions/min_length": 239.0, + "completions/max_length": 931.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 391.20001220703125, + "completions/min_terminated_length": 239.0, + "completions/max_terminated_length": 832.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5609374642372131, + "rewards/reward_func/std": 0.29702508449554443, + "reward": 0.5609374642372131, + "reward_std": 0.29702508449554443, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030493447557091713, + "sampling/sampling_logp_difference/max": 0.49188756942749023, + "sampling/importance_sampling_ratio/min": 0.2833422124385834, + "sampling/importance_sampling_ratio/mean": 1.0561542510986328, + "sampling/importance_sampling_ratio/max": 2.9562346935272217, + "kl": 0.007595470175147057, + "entropy": 0.06843653816031292, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.495979230850935, + "epoch": 0.00564453125, + "step": 289 + }, + { + "loss": 0.24644337594509125, + "grad_norm": 1.9443696737289429, + "learning_rate": 2.846153846153846e-07, + "num_tokens": 2858366.0, + "completions/mean_length": 593.625, + "completions/min_length": 255.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 487.3333435058594, + "completions/min_terminated_length": 255.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 14.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4854166805744171, + "rewards/reward_func/std": 0.28300929069519043, + "reward": 0.4854166805744171, + "reward_std": 0.28300929069519043, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002681639976799488, + "sampling/sampling_logp_difference/max": 1.0842843055725098, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.5773874521255493, + "sampling/importance_sampling_ratio/max": 1.164804458618164, + "kl": 0.006995630566962063, + "entropy": 0.05367319186916575, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.769579904153943, + "epoch": 0.0056640625, + "step": 290 + }, + { + "loss": -0.01919083297252655, + "grad_norm": 1.0992543697357178, + "learning_rate": 2.8205128205128203e-07, + "num_tokens": 2869319.0, + "completions/mean_length": 683.25, + "completions/min_length": 214.0, + "completions/max_length": 980.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 534.0, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 974.0, + "tools/call_frequency": 14.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3880208432674408, + "rewards/reward_func/std": 0.36770308017730713, + "reward": 0.3880208432674408, + "reward_std": 0.36770305037498474, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002354591153562069, + "sampling/sampling_logp_difference/max": 0.5601295828819275, + "sampling/importance_sampling_ratio/min": 0.16544069349765778, + "sampling/importance_sampling_ratio/mean": 0.6243107914924622, + "sampling/importance_sampling_ratio/max": 1.0807948112487793, + "kl": 0.005201722189667635, + "entropy": 0.040890268282964826, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.77484866976738, + "epoch": 0.00568359375, + "step": 291 + }, + { + "loss": 0.45926475524902344, + "grad_norm": 3.9946208000183105, + "learning_rate": 2.794871794871795e-07, + "num_tokens": 2878788.0, + "completions/mean_length": 499.375, + "completions/min_length": 209.0, + "completions/max_length": 935.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 440.5714416503906, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 935.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.46597224473953247, + "rewards/reward_func/std": 0.36586838960647583, + "reward": 0.46597224473953247, + "reward_std": 0.36586835980415344, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026659828145056963, + "sampling/sampling_logp_difference/max": 0.44251275062561035, + "sampling/importance_sampling_ratio/min": 0.5587342977523804, + "sampling/importance_sampling_ratio/mean": 1.092862844467163, + "sampling/importance_sampling_ratio/max": 1.8112249374389648, + "kl": 0.012606183037860319, + "entropy": 0.06572135421447456, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.666052618995309, + "epoch": 0.005703125, + "step": 292 + }, + { + "loss": 0.5182368159294128, + "grad_norm": 2.285550594329834, + "learning_rate": 2.7692307692307693e-07, + "num_tokens": 2887734.0, + "completions/mean_length": 433.5, + "completions/min_length": 209.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 269.16668701171875, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 343.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3295833468437195, + "rewards/reward_func/std": 0.307547926902771, + "reward": 0.3295833468437195, + "reward_std": 0.3075478971004486, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035625735763460398, + "sampling/sampling_logp_difference/max": 0.38818860054016113, + "sampling/importance_sampling_ratio/min": 0.20801369845867157, + "sampling/importance_sampling_ratio/mean": 0.8759403228759766, + "sampling/importance_sampling_ratio/max": 1.333734393119812, + "kl": 0.007428297147271223, + "entropy": 0.06528203457128257, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.498163321986794, + "epoch": 0.00572265625, + "step": 293 + }, + { + "loss": 0.36731377243995667, + "grad_norm": 2.333439826965332, + "learning_rate": 2.743589743589744e-07, + "num_tokens": 2897497.0, + "completions/mean_length": 534.625, + "completions/min_length": 244.0, + "completions/max_length": 944.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 413.3333435058594, + "completions/min_terminated_length": 244.0, + "completions/max_terminated_length": 944.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3140625059604645, + "rewards/reward_func/std": 0.2628629207611084, + "reward": 0.3140625059604645, + "reward_std": 0.2628629207611084, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003427007468417287, + "sampling/sampling_logp_difference/max": 0.43630897998809814, + "sampling/importance_sampling_ratio/min": 0.3076076805591583, + "sampling/importance_sampling_ratio/mean": 0.7473291754722595, + "sampling/importance_sampling_ratio/max": 1.5675846338272095, + "kl": 0.010240689865895547, + "entropy": 0.07025974185671657, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.841636132448912, + "epoch": 0.0057421875, + "step": 294 + }, + { + "loss": 0.17222344875335693, + "grad_norm": 2.3749332427978516, + "learning_rate": 2.7179487179487177e-07, + "num_tokens": 2905904.0, + "completions/mean_length": 364.25, + "completions/min_length": 232.0, + "completions/max_length": 942.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 364.25, + "completions/min_terminated_length": 232.0, + "completions/max_terminated_length": 942.0, + "tools/call_frequency": 8.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4600694179534912, + "rewards/reward_func/std": 0.17565974593162537, + "reward": 0.4600694179534912, + "reward_std": 0.17565973103046417, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005224711261689663, + "sampling/sampling_logp_difference/max": 0.7091436386108398, + "sampling/importance_sampling_ratio/min": 0.4167059361934662, + "sampling/importance_sampling_ratio/mean": 0.6987999677658081, + "sampling/importance_sampling_ratio/max": 0.9391834139823914, + "kl": 0.010752565940492786, + "entropy": 0.09412486385554075, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.06563300266862, + "epoch": 0.00576171875, + "step": 295 + }, + { + "loss": 0.16975122690200806, + "grad_norm": 3.536275625228882, + "learning_rate": 2.692307692307692e-07, + "num_tokens": 2914654.0, + "completions/mean_length": 409.375, + "completions/min_length": 176.0, + "completions/max_length": 899.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 252.83334350585938, + "completions/min_terminated_length": 176.0, + "completions/max_terminated_length": 334.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4229166805744171, + "rewards/reward_func/std": 0.36657872796058655, + "reward": 0.4229166805744171, + "reward_std": 0.36657872796058655, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0038839273620396852, + "sampling/sampling_logp_difference/max": 0.5424158573150635, + "sampling/importance_sampling_ratio/min": 0.4361511170864105, + "sampling/importance_sampling_ratio/mean": 0.9418432116508484, + "sampling/importance_sampling_ratio/max": 2.025289535522461, + "kl": 0.012379102714476176, + "entropy": 0.07638967945240438, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.862205067649484, + "epoch": 0.00578125, + "step": 296 + }, + { + "loss": 0.31052079796791077, + "grad_norm": 7.465968608856201, + "learning_rate": 2.6666666666666667e-07, + "num_tokens": 2924796.0, + "completions/mean_length": 583.125, + "completions/min_length": 237.0, + "completions/max_length": 922.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 471.0, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 917.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.38055557012557983, + "rewards/reward_func/std": 0.34188759326934814, + "reward": 0.38055557012557983, + "reward_std": 0.34188759326934814, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023708187509328127, + "sampling/sampling_logp_difference/max": 0.4755892753601074, + "sampling/importance_sampling_ratio/min": 0.598339855670929, + "sampling/importance_sampling_ratio/mean": 1.3391029834747314, + "sampling/importance_sampling_ratio/max": 2.2905690670013428, + "kl": 0.008296333078760654, + "entropy": 0.05683654185850173, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.014918126165867, + "epoch": 0.00580078125, + "step": 297 + }, + { + "loss": 0.9642153978347778, + "grad_norm": 7.006991386413574, + "learning_rate": 2.641025641025641e-07, + "num_tokens": 2934010.0, + "completions/mean_length": 465.75, + "completions/min_length": 169.0, + "completions/max_length": 922.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 400.5714416503906, + "completions/min_terminated_length": 169.0, + "completions/max_terminated_length": 885.0, + "tools/call_frequency": 11.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5792410969734192, + "rewards/reward_func/std": 0.3096855878829956, + "reward": 0.5792410969734192, + "reward_std": 0.3096855580806732, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002952637616544962, + "sampling/sampling_logp_difference/max": 0.4924130439758301, + "sampling/importance_sampling_ratio/min": 0.5188471674919128, + "sampling/importance_sampling_ratio/mean": 0.9176462888717651, + "sampling/importance_sampling_ratio/max": 2.184755802154541, + "kl": 0.013771501369774342, + "entropy": 0.04920901299919933, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.659406589344144, + "epoch": 0.0058203125, + "step": 298 + }, + { + "loss": 0.04990343004465103, + "grad_norm": 5.448063850402832, + "learning_rate": 2.615384615384615e-07, + "num_tokens": 2942044.0, + "completions/mean_length": 319.75, + "completions/min_length": 205.0, + "completions/max_length": 894.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 237.71429443359375, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 312.0, + "tools/call_frequency": 7.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6218750476837158, + "rewards/reward_func/std": 0.24459180235862732, + "reward": 0.6218750476837158, + "reward_std": 0.24459180235862732, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004642905667424202, + "sampling/sampling_logp_difference/max": 0.6759902238845825, + "sampling/importance_sampling_ratio/min": 0.2850757837295532, + "sampling/importance_sampling_ratio/mean": 1.0569398403167725, + "sampling/importance_sampling_ratio/max": 2.9483578205108643, + "kl": 0.01341090101050213, + "entropy": 0.0765160174923949, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.695970423519611, + "epoch": 0.00583984375, + "step": 299 + }, + { + "loss": -0.08633051812648773, + "grad_norm": 2.266727924346924, + "learning_rate": 2.5897435897435897e-07, + "num_tokens": 2952252.0, + "completions/mean_length": 591.125, + "completions/min_length": 236.0, + "completions/max_length": 922.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 403.0, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 917.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4077381193637848, + "rewards/reward_func/std": 0.35561978816986084, + "reward": 0.4077381193637848, + "reward_std": 0.35561978816986084, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002378544770181179, + "sampling/sampling_logp_difference/max": 0.490917444229126, + "sampling/importance_sampling_ratio/min": 0.3940471410751343, + "sampling/importance_sampling_ratio/mean": 0.8390909433364868, + "sampling/importance_sampling_ratio/max": 1.4945471286773682, + "kl": 0.0090175846562488, + "entropy": 0.052369412675034255, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.8701840788126, + "epoch": 0.005859375, + "step": 300 + }, + { + "loss": 0.12459854781627655, + "grad_norm": 4.095512390136719, + "learning_rate": 2.5641025641025636e-07, + "num_tokens": 2960405.0, + "completions/mean_length": 334.5, + "completions/min_length": 218.0, + "completions/max_length": 895.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 254.4285888671875, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 287.0, + "tools/call_frequency": 8.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.46059027314186096, + "rewards/reward_func/std": 0.3340311348438263, + "reward": 0.46059027314186096, + "reward_std": 0.3340311348438263, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.005092349834740162, + "sampling/sampling_logp_difference/max": 0.6349279880523682, + "sampling/importance_sampling_ratio/min": 0.5074770450592041, + "sampling/importance_sampling_ratio/mean": 0.907243549823761, + "sampling/importance_sampling_ratio/max": 1.4609812498092651, + "kl": 0.013518761494196951, + "entropy": 0.08455760549986735, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.949280824512243, + "epoch": 0.00587890625, + "step": 301 + }, + { + "loss": 0.33396098017692566, + "grad_norm": 1.9160410165786743, + "learning_rate": 2.538461538461538e-07, + "num_tokens": 2971289.0, + "completions/mean_length": 674.5, + "completions/min_length": 272.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 540.2000122070312, + "completions/min_terminated_length": 272.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.43281251192092896, + "rewards/reward_func/std": 0.3507251441478729, + "reward": 0.43281251192092896, + "reward_std": 0.3507251441478729, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024716113694012165, + "sampling/sampling_logp_difference/max": 0.7191798686981201, + "sampling/importance_sampling_ratio/min": 0.41691187024116516, + "sampling/importance_sampling_ratio/mean": 0.80147385597229, + "sampling/importance_sampling_ratio/max": 1.1808468103408813, + "kl": 0.007806519191944972, + "entropy": 0.046461939899018034, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.282068559899926, + "epoch": 0.0058984375, + "step": 302 + }, + { + "loss": 0.4103955328464508, + "grad_norm": 3.2776429653167725, + "learning_rate": 2.5128205128205126e-07, + "num_tokens": 2980296.0, + "completions/mean_length": 440.125, + "completions/min_length": 208.0, + "completions/max_length": 946.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 373.0000305175781, + "completions/min_terminated_length": 208.0, + "completions/max_terminated_length": 946.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.41961807012557983, + "rewards/reward_func/std": 0.27931660413742065, + "reward": 0.41961807012557983, + "reward_std": 0.27931660413742065, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004904399625957012, + "sampling/sampling_logp_difference/max": 0.9670320153236389, + "sampling/importance_sampling_ratio/min": 0.28145715594291687, + "sampling/importance_sampling_ratio/mean": 0.9721834659576416, + "sampling/importance_sampling_ratio/max": 2.0933451652526855, + "kl": 0.008778297087701503, + "entropy": 0.08727305883076042, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.376354286447167, + "epoch": 0.00591796875, + "step": 303 + }, + { + "loss": 0.6357797384262085, + "grad_norm": 4.039347171783447, + "learning_rate": 2.487179487179487e-07, + "num_tokens": 2989098.0, + "completions/mean_length": 415.125, + "completions/min_length": 210.0, + "completions/max_length": 937.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 340.5714416503906, + "completions/min_terminated_length": 210.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 9.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5293750166893005, + "rewards/reward_func/std": 0.14520865678787231, + "reward": 0.5293750166893005, + "reward_std": 0.1452086716890335, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034858037251979113, + "sampling/sampling_logp_difference/max": 0.5126770734786987, + "sampling/importance_sampling_ratio/min": 0.5372920632362366, + "sampling/importance_sampling_ratio/mean": 1.1084141731262207, + "sampling/importance_sampling_ratio/max": 1.9726964235305786, + "kl": 0.01047392409236636, + "entropy": 0.063438531011343, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.402306104078889, + "epoch": 0.0059375, + "step": 304 + }, + { + "loss": 0.484939843416214, + "grad_norm": 3.2674334049224854, + "learning_rate": 2.4615384615384616e-07, + "num_tokens": 2998886.0, + "completions/mean_length": 539.125, + "completions/min_length": 209.0, + "completions/max_length": 945.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 309.0, + "completions/min_terminated_length": 209.0, + "completions/max_terminated_length": 517.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4564772844314575, + "rewards/reward_func/std": 0.2492385059595108, + "reward": 0.4564772844314575, + "reward_std": 0.2492384910583496, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003891421016305685, + "sampling/sampling_logp_difference/max": 0.6988792419433594, + "sampling/importance_sampling_ratio/min": 0.31638389825820923, + "sampling/importance_sampling_ratio/mean": 0.8427509069442749, + "sampling/importance_sampling_ratio/max": 1.5236088037490845, + "kl": 0.006430305278627202, + "entropy": 0.07659899146528915, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.623323649168015, + "epoch": 0.00595703125, + "step": 305 + }, + { + "loss": 0.6164342761039734, + "grad_norm": 3.31184720993042, + "learning_rate": 2.4358974358974356e-07, + "num_tokens": 3007608.0, + "completions/mean_length": 404.875, + "completions/min_length": 190.0, + "completions/max_length": 935.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 228.83334350585938, + "completions/min_terminated_length": 190.0, + "completions/max_terminated_length": 272.0, + "tools/call_frequency": 9.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47083333134651184, + "rewards/reward_func/std": 0.34985116124153137, + "reward": 0.47083333134651184, + "reward_std": 0.349851131439209, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0028322956059128046, + "sampling/sampling_logp_difference/max": 0.40680623054504395, + "sampling/importance_sampling_ratio/min": 0.5165644288063049, + "sampling/importance_sampling_ratio/mean": 1.0549507141113281, + "sampling/importance_sampling_ratio/max": 1.7601639032363892, + "kl": 0.011312763439491391, + "entropy": 0.06436907115858048, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.98656165227294, + "epoch": 0.0059765625, + "step": 306 + }, + { + "loss": -0.18740525841712952, + "grad_norm": 2.970656633377075, + "learning_rate": 2.41025641025641e-07, + "num_tokens": 3016975.0, + "completions/mean_length": 485.125, + "completions/min_length": 44.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 418.14288330078125, + "completions/min_terminated_length": 44.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 11.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.34375, + "rewards/reward_func/std": 0.358319491147995, + "reward": 0.34375, + "reward_std": 0.358319491147995, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003335353685542941, + "sampling/sampling_logp_difference/max": 0.4514467716217041, + "sampling/importance_sampling_ratio/min": 0.18746522068977356, + "sampling/importance_sampling_ratio/mean": 1.122767686843872, + "sampling/importance_sampling_ratio/max": 2.711134433746338, + "kl": 0.007319065276533365, + "entropy": 0.07329587155254558, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.97095856629312, + "epoch": 0.00599609375, + "step": 307 + }, + { + "loss": 0.598524808883667, + "grad_norm": 4.890788555145264, + "learning_rate": 2.3846153846153846e-07, + "num_tokens": 3026716.0, + "completions/mean_length": 531.5, + "completions/min_length": 224.0, + "completions/max_length": 949.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 397.8333435058594, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35381942987442017, + "rewards/reward_func/std": 0.22382698953151703, + "reward": 0.35381942987442017, + "reward_std": 0.22382697463035583, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004458331502974033, + "sampling/sampling_logp_difference/max": 0.7083139419555664, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.9891330003738403, + "sampling/importance_sampling_ratio/max": 1.9646192789077759, + "kl": 0.008309213466418441, + "entropy": 0.08776952675543725, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.411638690158725, + "epoch": 0.006015625, + "step": 308 + }, + { + "loss": -0.09724443405866623, + "grad_norm": 4.801892280578613, + "learning_rate": 2.3589743589743588e-07, + "num_tokens": 3034333.0, + "completions/mean_length": 267.625, + "completions/min_length": 221.0, + "completions/max_length": 329.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 267.625, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 329.0, + "tools/call_frequency": 6.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5369296073913574, + "rewards/reward_func/std": 0.19295206665992737, + "reward": 0.5369296073913574, + "reward_std": 0.19295206665992737, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.008726969361305237, + "sampling/sampling_logp_difference/max": 0.7301063537597656, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.7749390602111816, + "sampling/importance_sampling_ratio/max": 1.7247945070266724, + "kl": 0.012253313034307212, + "entropy": 0.10526352189481258, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.449104344472289, + "epoch": 0.00603515625, + "step": 309 + }, + { + "loss": 0.5771793127059937, + "grad_norm": 2.772547721862793, + "learning_rate": 2.3333333333333333e-07, + "num_tokens": 3043721.0, + "completions/mean_length": 488.25, + "completions/min_length": 218.0, + "completions/max_length": 905.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 488.25, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 905.0, + "tools/call_frequency": 11.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6885457038879395, + "rewards/reward_func/std": 0.30560970306396484, + "reward": 0.6885457038879395, + "reward_std": 0.30560970306396484, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002327671740204096, + "sampling/sampling_logp_difference/max": 0.3663747310638428, + "sampling/importance_sampling_ratio/min": 0.5477707982063293, + "sampling/importance_sampling_ratio/mean": 1.2031095027923584, + "sampling/importance_sampling_ratio/max": 1.749635934829712, + "kl": 0.0074971493959310465, + "entropy": 0.05789940431714058, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.45246941037476, + "epoch": 0.0060546875, + "step": 310 + }, + { + "loss": 0.2967388331890106, + "grad_norm": 3.4299941062927246, + "learning_rate": 2.3076923076923078e-07, + "num_tokens": 3051889.0, + "completions/mean_length": 334.875, + "completions/min_length": 202.0, + "completions/max_length": 901.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 254.00001525878906, + "completions/min_terminated_length": 202.0, + "completions/max_terminated_length": 323.0, + "tools/call_frequency": 7.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5764912366867065, + "rewards/reward_func/std": 0.2964637577533722, + "reward": 0.5764912366867065, + "reward_std": 0.2964637577533722, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006043317727744579, + "sampling/sampling_logp_difference/max": 0.6358933448791504, + "sampling/importance_sampling_ratio/min": 0.17751385271549225, + "sampling/importance_sampling_ratio/mean": 0.7416783571243286, + "sampling/importance_sampling_ratio/max": 1.649349331855774, + "kl": 0.012622155423741788, + "entropy": 0.09290892747230828, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.21427883207798, + "epoch": 0.00607421875, + "step": 311 + }, + { + "loss": 0.5077472925186157, + "grad_norm": 2.5424463748931885, + "learning_rate": 2.2820512820512818e-07, + "num_tokens": 3061458.0, + "completions/mean_length": 511.375, + "completions/min_length": 175.0, + "completions/max_length": 960.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 453.2857360839844, + "completions/min_terminated_length": 175.0, + "completions/max_terminated_length": 960.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4880208373069763, + "rewards/reward_func/std": 0.3518618047237396, + "reward": 0.4880208373069763, + "reward_std": 0.3518618047237396, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003838771255686879, + "sampling/sampling_logp_difference/max": 0.5969632863998413, + "sampling/importance_sampling_ratio/min": 0.4891752004623413, + "sampling/importance_sampling_ratio/mean": 0.9562264680862427, + "sampling/importance_sampling_ratio/max": 1.8094364404678345, + "kl": 0.00837391991080949, + "entropy": 0.07098359445808455, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.228636862710118, + "epoch": 0.00609375, + "step": 312 + }, + { + "loss": 0.10866241157054901, + "grad_norm": 3.127600908279419, + "learning_rate": 2.2564102564102563e-07, + "num_tokens": 3070880.0, + "completions/mean_length": 493.5, + "completions/min_length": 190.0, + "completions/max_length": 931.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 431.0000305175781, + "completions/min_terminated_length": 190.0, + "completions/max_terminated_length": 928.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4663541615009308, + "rewards/reward_func/std": 0.3487545847892761, + "reward": 0.4663541615009308, + "reward_std": 0.34875455498695374, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0032199728302657604, + "sampling/sampling_logp_difference/max": 0.4863290786743164, + "sampling/importance_sampling_ratio/min": 0.5391635298728943, + "sampling/importance_sampling_ratio/mean": 0.8321606516838074, + "sampling/importance_sampling_ratio/max": 1.4861879348754883, + "kl": 0.01275488109968137, + "entropy": 0.07472763175610453, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.389009600505233, + "epoch": 0.00611328125, + "step": 313 + }, + { + "loss": -0.33621665835380554, + "grad_norm": 6.640406131744385, + "learning_rate": 2.2307692307692308e-07, + "num_tokens": 3078384.0, + "completions/mean_length": 251.625, + "completions/min_length": 203.0, + "completions/max_length": 311.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 251.625, + "completions/min_terminated_length": 203.0, + "completions/max_terminated_length": 311.0, + "tools/call_frequency": 6.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6257292032241821, + "rewards/reward_func/std": 0.2595895826816559, + "reward": 0.6257292032241821, + "reward_std": 0.2595895826816559, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00792793557047844, + "sampling/sampling_logp_difference/max": 0.6692848205566406, + "sampling/importance_sampling_ratio/min": 0.1932678073644638, + "sampling/importance_sampling_ratio/mean": 1.20702064037323, + "sampling/importance_sampling_ratio/max": 2.579185962677002, + "kl": 0.010588183533400297, + "entropy": 0.08632633835077286, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.513603920117021, + "epoch": 0.0061328125, + "step": 314 + }, + { + "loss": -0.08105252683162689, + "grad_norm": 6.729301452636719, + "learning_rate": 2.205128205128205e-07, + "num_tokens": 3087256.0, + "completions/mean_length": 423.25, + "completions/min_length": 224.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 358.71429443359375, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 938.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5800978541374207, + "rewards/reward_func/std": 0.2640564739704132, + "reward": 0.5800978541374207, + "reward_std": 0.2640564739704132, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004214680287986994, + "sampling/sampling_logp_difference/max": 1.1795425415039062, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.8384581804275513, + "sampling/importance_sampling_ratio/max": 1.9708654880523682, + "kl": 0.012273009691853076, + "entropy": 0.07651870994595811, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.508617043495178, + "epoch": 0.00615234375, + "step": 315 + }, + { + "loss": 0.10263785719871521, + "grad_norm": 3.5244534015655518, + "learning_rate": 2.1794871794871795e-07, + "num_tokens": 3096172.0, + "completions/mean_length": 428.375, + "completions/min_length": 189.0, + "completions/max_length": 934.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 261.3333435058594, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 329.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4234375059604645, + "rewards/reward_func/std": 0.191795215010643, + "reward": 0.4234375059604645, + "reward_std": 0.191795215010643, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004551462363451719, + "sampling/sampling_logp_difference/max": 0.4883997440338135, + "sampling/importance_sampling_ratio/min": 0.4289291799068451, + "sampling/importance_sampling_ratio/mean": 1.2554824352264404, + "sampling/importance_sampling_ratio/max": 1.9863964319229126, + "kl": 0.013009324262384325, + "entropy": 0.08144828182412311, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.039322705939412, + "epoch": 0.006171875, + "step": 316 + }, + { + "loss": 0.3706982433795929, + "grad_norm": 3.3316924571990967, + "learning_rate": 2.153846153846154e-07, + "num_tokens": 3105772.0, + "completions/mean_length": 513.75, + "completions/min_length": 227.0, + "completions/max_length": 983.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 365.0, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 906.0, + "tools/call_frequency": 11.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.42847222089767456, + "rewards/reward_func/std": 0.27175918221473694, + "reward": 0.42847222089767456, + "reward_std": 0.27175918221473694, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00287108076736331, + "sampling/sampling_logp_difference/max": 0.4538072347640991, + "sampling/importance_sampling_ratio/min": 0.12980644404888153, + "sampling/importance_sampling_ratio/mean": 1.075140357017517, + "sampling/importance_sampling_ratio/max": 1.940106749534607, + "kl": 0.00755229163041804, + "entropy": 0.06822676910087466, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.720922349020839, + "epoch": 0.00619140625, + "step": 317 + }, + { + "loss": 0.055338695645332336, + "grad_norm": 3.1102559566497803, + "learning_rate": 2.128205128205128e-07, + "num_tokens": 3116619.0, + "completions/mean_length": 670.375, + "completions/min_length": 243.0, + "completions/max_length": 935.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 429.0, + "completions/min_terminated_length": 243.0, + "completions/max_terminated_length": 873.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.37881946563720703, + "rewards/reward_func/std": 0.2847727835178375, + "reward": 0.37881946563720703, + "reward_std": 0.2847727835178375, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002331752562895417, + "sampling/sampling_logp_difference/max": 0.6193728446960449, + "sampling/importance_sampling_ratio/min": 0.3196961581707001, + "sampling/importance_sampling_ratio/mean": 1.175635576248169, + "sampling/importance_sampling_ratio/max": 1.9433002471923828, + "kl": 0.005369035206967965, + "entropy": 0.04382790718227625, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.589711416512728, + "epoch": 0.0062109375, + "step": 318 + }, + { + "loss": 0.3791399598121643, + "grad_norm": 1.9142252206802368, + "learning_rate": 2.1025641025641025e-07, + "num_tokens": 3126078.0, + "completions/mean_length": 497.125, + "completions/min_length": 222.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 367.8333435058594, + "completions/min_terminated_length": 222.0, + "completions/max_terminated_length": 929.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4374256134033203, + "rewards/reward_func/std": 0.3344995081424713, + "reward": 0.4374256134033203, + "reward_std": 0.3344995081424713, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0041219755075871944, + "sampling/sampling_logp_difference/max": 0.4924802780151367, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6251306533813477, + "sampling/importance_sampling_ratio/max": 1.3578404188156128, + "kl": 0.009573580959113315, + "entropy": 0.06549935310613364, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.948549998924136, + "epoch": 0.00623046875, + "step": 319 + }, + { + "loss": 0.1375473141670227, + "grad_norm": 2.716317892074585, + "learning_rate": 2.076923076923077e-07, + "num_tokens": 3136238.0, + "completions/mean_length": 584.5, + "completions/min_length": 237.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 386.20001220703125, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 899.0, + "tools/call_frequency": 13.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3654513955116272, + "rewards/reward_func/std": 0.29816684126853943, + "reward": 0.3654513955116272, + "reward_std": 0.29816684126853943, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002822891576215625, + "sampling/sampling_logp_difference/max": 0.4751272201538086, + "sampling/importance_sampling_ratio/min": 0.4323903024196625, + "sampling/importance_sampling_ratio/mean": 0.9953634738922119, + "sampling/importance_sampling_ratio/max": 1.7258116006851196, + "kl": 0.009171866109682014, + "entropy": 0.07099261204712093, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.67844346538186, + "epoch": 0.00625, + "step": 320 + }, + { + "loss": 0.26897934079170227, + "grad_norm": 5.106388568878174, + "learning_rate": 2.0512820512820512e-07, + "num_tokens": 3144386.0, + "completions/mean_length": 333.5, + "completions/min_length": 219.0, + "completions/max_length": 925.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 333.5, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 925.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.49899303913116455, + "rewards/reward_func/std": 0.2577325999736786, + "reward": 0.49899303913116455, + "reward_std": 0.2577325999736786, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00509037496522069, + "sampling/sampling_logp_difference/max": 0.6850378513336182, + "sampling/importance_sampling_ratio/min": 0.3920983076095581, + "sampling/importance_sampling_ratio/mean": 0.7732139825820923, + "sampling/importance_sampling_ratio/max": 1.3786165714263916, + "kl": 0.01056574282119982, + "entropy": 0.08332786016399041, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.9476088732481, + "epoch": 0.00626953125, + "step": 321 + }, + { + "loss": -0.37858039140701294, + "grad_norm": 2.267113447189331, + "learning_rate": 2.0256410256410257e-07, + "num_tokens": 3153292.0, + "completions/mean_length": 428.625, + "completions/min_length": 217.0, + "completions/max_length": 909.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 360.0000305175781, + "completions/min_terminated_length": 217.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5850297212600708, + "rewards/reward_func/std": 0.32072994112968445, + "reward": 0.5850297212600708, + "reward_std": 0.32072994112968445, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0039230757392942905, + "sampling/sampling_logp_difference/max": 0.40785539150238037, + "sampling/importance_sampling_ratio/min": 0.4799075722694397, + "sampling/importance_sampling_ratio/mean": 0.9917155504226685, + "sampling/importance_sampling_ratio/max": 1.9045662879943848, + "kl": 0.010754496994195506, + "entropy": 0.07850070181302726, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.759819220751524, + "epoch": 0.0062890625, + "step": 322 + }, + { + "loss": 0.17047777771949768, + "grad_norm": 3.2805025577545166, + "learning_rate": 2e-07, + "num_tokens": 3162976.0, + "completions/mean_length": 524.875, + "completions/min_length": 252.0, + "completions/max_length": 951.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 390.66668701171875, + "completions/min_terminated_length": 252.0, + "completions/max_terminated_length": 900.0, + "tools/call_frequency": 12.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.43802082538604736, + "rewards/reward_func/std": 0.25310999155044556, + "reward": 0.43802082538604736, + "reward_std": 0.25310999155044556, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0032428160775452852, + "sampling/sampling_logp_difference/max": 0.48598766326904297, + "sampling/importance_sampling_ratio/min": 0.4553104341030121, + "sampling/importance_sampling_ratio/mean": 0.9057520031929016, + "sampling/importance_sampling_ratio/max": 1.618770956993103, + "kl": 0.0059548512363107875, + "entropy": 0.054720557935070246, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.49704859405756, + "epoch": 0.00630859375, + "step": 323 + }, + { + "loss": 0.22346168756484985, + "grad_norm": 2.197716474533081, + "learning_rate": 1.9743589743589741e-07, + "num_tokens": 3171332.0, + "completions/mean_length": 359.125, + "completions/min_length": 210.0, + "completions/max_length": 916.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 279.5714416503906, + "completions/min_terminated_length": 210.0, + "completions/max_terminated_length": 348.0, + "tools/call_frequency": 8.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5092644691467285, + "rewards/reward_func/std": 0.35833778977394104, + "reward": 0.5092644691467285, + "reward_std": 0.35833775997161865, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.007166056893765926, + "sampling/sampling_logp_difference/max": 1.6195255517959595, + "sampling/importance_sampling_ratio/min": 0.06338764727115631, + "sampling/importance_sampling_ratio/mean": 0.6000833511352539, + "sampling/importance_sampling_ratio/max": 1.7804350852966309, + "kl": 0.01843526231823489, + "entropy": 0.09932832047343254, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.25578748434782, + "epoch": 0.006328125, + "step": 324 + }, + { + "loss": 0.18812786042690277, + "grad_norm": 2.756762742996216, + "learning_rate": 1.9487179487179486e-07, + "num_tokens": 3182795.0, + "completions/mean_length": 748.25, + "completions/min_length": 207.0, + "completions/max_length": 942.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 587.5, + "completions/min_terminated_length": 207.0, + "completions/max_terminated_length": 942.0, + "tools/call_frequency": 17.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.2923200726509094, + "rewards/reward_func/std": 0.3412052094936371, + "reward": 0.2923200726509094, + "reward_std": 0.3412052094936371, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0013113220920786262, + "sampling/sampling_logp_difference/max": 0.4617457389831543, + "sampling/importance_sampling_ratio/min": 0.31652429699897766, + "sampling/importance_sampling_ratio/mean": 0.7205790281295776, + "sampling/importance_sampling_ratio/max": 1.1665351390838623, + "kl": 0.007424278403050266, + "entropy": 0.026594726485200226, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.81009803339839, + "epoch": 0.00634765625, + "step": 325 + }, + { + "loss": -0.4382976293563843, + "grad_norm": 3.1490488052368164, + "learning_rate": 1.9230769230769231e-07, + "num_tokens": 3191618.0, + "completions/mean_length": 417.75, + "completions/min_length": 230.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 262.16668701171875, + "completions/min_terminated_length": 230.0, + "completions/max_terminated_length": 347.0, + "tools/call_frequency": 10.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6100297570228577, + "rewards/reward_func/std": 0.20313984155654907, + "reward": 0.6100297570228577, + "reward_std": 0.20313984155654907, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0038444146048277617, + "sampling/sampling_logp_difference/max": 0.42541563510894775, + "sampling/importance_sampling_ratio/min": 0.3855036497116089, + "sampling/importance_sampling_ratio/mean": 0.8040324449539185, + "sampling/importance_sampling_ratio/max": 1.2527273893356323, + "kl": 0.012028418539557606, + "entropy": 0.08702858758624643, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.347337443381548, + "epoch": 0.0063671875, + "step": 326 + }, + { + "loss": 0.6224079728126526, + "grad_norm": 3.31811785697937, + "learning_rate": 1.8974358974358974e-07, + "num_tokens": 3200484.0, + "completions/mean_length": 422.75, + "completions/min_length": 185.0, + "completions/max_length": 949.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 355.0000305175781, + "completions/min_terminated_length": 185.0, + "completions/max_terminated_length": 949.0, + "tools/call_frequency": 9.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.43513256311416626, + "rewards/reward_func/std": 0.32992058992385864, + "reward": 0.43513256311416626, + "reward_std": 0.32992058992385864, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004206390120089054, + "sampling/sampling_logp_difference/max": 0.48552119731903076, + "sampling/importance_sampling_ratio/min": 0.34344032406806946, + "sampling/importance_sampling_ratio/mean": 0.871833086013794, + "sampling/importance_sampling_ratio/max": 1.408193588256836, + "kl": 0.009240949875675142, + "entropy": 0.07529064960544929, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.557215774431825, + "epoch": 0.00638671875, + "step": 327 + }, + { + "loss": 0.003911484964191914, + "grad_norm": 1.2822294235229492, + "learning_rate": 1.8717948717948716e-07, + "num_tokens": 3210725.0, + "completions/mean_length": 594.25, + "completions/min_length": 221.0, + "completions/max_length": 991.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 543.2857666015625, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 991.0, + "tools/call_frequency": 13.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.569444477558136, + "rewards/reward_func/std": 0.24764879047870636, + "reward": 0.569444477558136, + "reward_std": 0.24764879047870636, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025004958733916283, + "sampling/sampling_logp_difference/max": 0.660049557685852, + "sampling/importance_sampling_ratio/min": 0.3581734895706177, + "sampling/importance_sampling_ratio/mean": 0.753393828868866, + "sampling/importance_sampling_ratio/max": 1.3661167621612549, + "kl": 0.006235106411622837, + "entropy": 0.0466050315881148, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.79836393892765, + "epoch": 0.00640625, + "step": 328 + }, + { + "loss": 0.4997605085372925, + "grad_norm": 1.5005872249603271, + "learning_rate": 1.846153846153846e-07, + "num_tokens": 3220908.0, + "completions/mean_length": 588.125, + "completions/min_length": 211.0, + "completions/max_length": 979.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 386.6000061035156, + "completions/min_terminated_length": 211.0, + "completions/max_terminated_length": 907.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33348482847213745, + "rewards/reward_func/std": 0.31152936816215515, + "reward": 0.33348482847213745, + "reward_std": 0.31152936816215515, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026560677215456963, + "sampling/sampling_logp_difference/max": 0.5855637788772583, + "sampling/importance_sampling_ratio/min": 0.4338342845439911, + "sampling/importance_sampling_ratio/mean": 0.7634598016738892, + "sampling/importance_sampling_ratio/max": 1.351252794265747, + "kl": 0.009469829441513866, + "entropy": 0.05527725757565349, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.414971269667149, + "epoch": 0.00642578125, + "step": 329 + }, + { + "loss": 0.5915253162384033, + "grad_norm": 2.9866275787353516, + "learning_rate": 1.8205128205128203e-07, + "num_tokens": 3229845.0, + "completions/mean_length": 432.5, + "completions/min_length": 189.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 432.5, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 953.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4713541567325592, + "rewards/reward_func/std": 0.2592479884624481, + "reward": 0.4713541567325592, + "reward_std": 0.25924795866012573, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0037110636476427317, + "sampling/sampling_logp_difference/max": 0.4554119110107422, + "sampling/importance_sampling_ratio/min": 0.59665846824646, + "sampling/importance_sampling_ratio/mean": 1.105031967163086, + "sampling/importance_sampling_ratio/max": 2.3276426792144775, + "kl": 0.006527607649331912, + "entropy": 0.06372938037384301, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.420138906687498, + "epoch": 0.0064453125, + "step": 330 + }, + { + "loss": 0.05264908820390701, + "grad_norm": 1.3418209552764893, + "learning_rate": 1.7948717948717948e-07, + "num_tokens": 3240499.0, + "completions/mean_length": 647.375, + "completions/min_length": 240.0, + "completions/max_length": 913.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 513.0, + "completions/min_terminated_length": 240.0, + "completions/max_terminated_length": 913.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5519444346427917, + "rewards/reward_func/std": 0.3300870954990387, + "reward": 0.5519444346427917, + "reward_std": 0.3300870954990387, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002090171677991748, + "sampling/sampling_logp_difference/max": 0.7620921730995178, + "sampling/importance_sampling_ratio/min": 0.2942596971988678, + "sampling/importance_sampling_ratio/mean": 0.7811707258224487, + "sampling/importance_sampling_ratio/max": 1.2021929025650024, + "kl": 0.0062578407232649624, + "entropy": 0.05178303591674194, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.913652652874589, + "epoch": 0.00646484375, + "step": 331 + }, + { + "loss": 0.09823893755674362, + "grad_norm": 3.1183011531829834, + "learning_rate": 1.7692307692307693e-07, + "num_tokens": 3249979.0, + "completions/mean_length": 499.375, + "completions/min_length": 219.0, + "completions/max_length": 912.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 366.66668701171875, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 912.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5562499761581421, + "rewards/reward_func/std": 0.3278038501739502, + "reward": 0.5562499761581421, + "reward_std": 0.3278038501739502, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002376953372731805, + "sampling/sampling_logp_difference/max": 0.3333643674850464, + "sampling/importance_sampling_ratio/min": 0.6394515633583069, + "sampling/importance_sampling_ratio/mean": 1.1774482727050781, + "sampling/importance_sampling_ratio/max": 1.8985029458999634, + "kl": 0.005948984413407743, + "entropy": 0.05854130076477304, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.535399669781327, + "epoch": 0.006484375, + "step": 332 + }, + { + "loss": 0.17652484774589539, + "grad_norm": 3.3799846172332764, + "learning_rate": 1.7435897435897435e-07, + "num_tokens": 3259543.0, + "completions/mean_length": 511.25, + "completions/min_length": 195.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 272.8000183105469, + "completions/min_terminated_length": 195.0, + "completions/max_terminated_length": 361.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44218748807907104, + "rewards/reward_func/std": 0.31919756531715393, + "reward": 0.44218748807907104, + "reward_std": 0.31919756531715393, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003928401507437229, + "sampling/sampling_logp_difference/max": 0.7377955913543701, + "sampling/importance_sampling_ratio/min": 0.3710220158100128, + "sampling/importance_sampling_ratio/mean": 0.998168408870697, + "sampling/importance_sampling_ratio/max": 1.6550899744033813, + "kl": 0.00888295695767738, + "entropy": 0.07552083872724324, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.498794274404645, + "epoch": 0.00650390625, + "step": 333 + }, + { + "loss": 0.8631119728088379, + "grad_norm": 6.452493190765381, + "learning_rate": 1.7179487179487178e-07, + "num_tokens": 3269011.0, + "completions/mean_length": 498.375, + "completions/min_length": 213.0, + "completions/max_length": 904.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 376.8333435058594, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 904.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5817708373069763, + "rewards/reward_func/std": 0.27731654047966003, + "reward": 0.5817708373069763, + "reward_std": 0.27731654047966003, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002825594274327159, + "sampling/sampling_logp_difference/max": 0.5126772522926331, + "sampling/importance_sampling_ratio/min": 0.6094498634338379, + "sampling/importance_sampling_ratio/mean": 1.1836626529693604, + "sampling/importance_sampling_ratio/max": 2.5954129695892334, + "kl": 0.008416564262006432, + "entropy": 0.05085090681677684, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.28621200285852, + "epoch": 0.0065234375, + "step": 334 + }, + { + "loss": 0.3602752089500427, + "grad_norm": 3.0999910831451416, + "learning_rate": 1.6923076923076923e-07, + "num_tokens": 3278521.0, + "completions/mean_length": 503.25, + "completions/min_length": 177.0, + "completions/max_length": 940.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 440.857177734375, + "completions/min_terminated_length": 177.0, + "completions/max_terminated_length": 900.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.31041666865348816, + "rewards/reward_func/std": 0.33187180757522583, + "reward": 0.31041666865348816, + "reward_std": 0.33187177777290344, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035550384782254696, + "sampling/sampling_logp_difference/max": 0.4760756492614746, + "sampling/importance_sampling_ratio/min": 0.29481732845306396, + "sampling/importance_sampling_ratio/mean": 0.9914381504058838, + "sampling/importance_sampling_ratio/max": 1.7381129264831543, + "kl": 0.013805479102302343, + "entropy": 0.068711306899786, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.913112973794341, + "epoch": 0.00654296875, + "step": 335 + }, + { + "loss": -0.13825899362564087, + "grad_norm": 2.51481556892395, + "learning_rate": 1.6666666666666665e-07, + "num_tokens": 3287994.0, + "completions/mean_length": 498.625, + "completions/min_length": 221.0, + "completions/max_length": 955.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 261.20001220703125, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 293.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5489583015441895, + "rewards/reward_func/std": 0.1640380471944809, + "reward": 0.5489583015441895, + "reward_std": 0.1640380471944809, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003267665859311819, + "sampling/sampling_logp_difference/max": 0.4715311527252197, + "sampling/importance_sampling_ratio/min": 0.5103208422660828, + "sampling/importance_sampling_ratio/mean": 1.2859299182891846, + "sampling/importance_sampling_ratio/max": 2.381342649459839, + "kl": 0.006717846103128977, + "entropy": 0.06065139168640599, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.603814678266644, + "epoch": 0.0065625, + "step": 336 + }, + { + "loss": 0.7182648181915283, + "grad_norm": 2.1415321826934814, + "learning_rate": 1.641025641025641e-07, + "num_tokens": 3298186.0, + "completions/mean_length": 588.875, + "completions/min_length": 218.0, + "completions/max_length": 969.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 482.16668701171875, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 969.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.39318811893463135, + "rewards/reward_func/std": 0.37508052587509155, + "reward": 0.39318811893463135, + "reward_std": 0.37508052587509155, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019707351457327604, + "sampling/sampling_logp_difference/max": 0.25018590688705444, + "sampling/importance_sampling_ratio/min": 0.7164834141731262, + "sampling/importance_sampling_ratio/mean": 1.125996470451355, + "sampling/importance_sampling_ratio/max": 1.8753738403320312, + "kl": 0.007458932290319353, + "entropy": 0.05514801945537329, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.009446874260902, + "epoch": 0.00658203125, + "step": 337 + }, + { + "loss": 0.5119581818580627, + "grad_norm": 3.3310132026672363, + "learning_rate": 1.6153846153846155e-07, + "num_tokens": 3307687.0, + "completions/mean_length": 502.5, + "completions/min_length": 210.0, + "completions/max_length": 957.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 437.5714416503906, + "completions/min_terminated_length": 210.0, + "completions/max_terminated_length": 930.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4619791805744171, + "rewards/reward_func/std": 0.35454419255256653, + "reward": 0.4619791805744171, + "reward_std": 0.35454419255256653, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002755816327407956, + "sampling/sampling_logp_difference/max": 0.4759345054626465, + "sampling/importance_sampling_ratio/min": 0.4779449701309204, + "sampling/importance_sampling_ratio/mean": 0.9571847915649414, + "sampling/importance_sampling_ratio/max": 1.6773992776870728, + "kl": 0.008574052480980754, + "entropy": 0.05683722865069285, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.68752052821219, + "epoch": 0.0066015625, + "step": 338 + }, + { + "loss": -0.409617155790329, + "grad_norm": 1.8384442329406738, + "learning_rate": 1.5897435897435895e-07, + "num_tokens": 3316461.0, + "completions/mean_length": 411.375, + "completions/min_length": 186.0, + "completions/max_length": 897.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 249.5, + "completions/min_terminated_length": 186.0, + "completions/max_terminated_length": 289.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5638731122016907, + "rewards/reward_func/std": 0.3237993121147156, + "reward": 0.5638731122016907, + "reward_std": 0.3237993121147156, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002885414520278573, + "sampling/sampling_logp_difference/max": 0.5216432213783264, + "sampling/importance_sampling_ratio/min": 0.39512306451797485, + "sampling/importance_sampling_ratio/mean": 0.9706514477729797, + "sampling/importance_sampling_ratio/max": 1.469448208808899, + "kl": 0.007979535614140332, + "entropy": 0.0671772378263995, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.902006983757019, + "epoch": 0.00662109375, + "step": 339 + }, + { + "loss": 0.16340669989585876, + "grad_norm": 2.3925621509552, + "learning_rate": 1.564102564102564e-07, + "num_tokens": 3326584.0, + "completions/mean_length": 580.25, + "completions/min_length": 226.0, + "completions/max_length": 982.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 368.8000183105469, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 884.0, + "tools/call_frequency": 13.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4592013955116272, + "rewards/reward_func/std": 0.3683664798736572, + "reward": 0.4592013955116272, + "reward_std": 0.3683664798736572, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00206647627055645, + "sampling/sampling_logp_difference/max": 0.6098775863647461, + "sampling/importance_sampling_ratio/min": 0.6060183048248291, + "sampling/importance_sampling_ratio/mean": 1.0016295909881592, + "sampling/importance_sampling_ratio/max": 1.7806556224822998, + "kl": 0.005329208779585315, + "entropy": 0.05586054385639727, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.59595831669867, + "epoch": 0.006640625, + "step": 340 + }, + { + "loss": 0.15290866792201996, + "grad_norm": 2.6778690814971924, + "learning_rate": 1.5384615384615385e-07, + "num_tokens": 3336802.0, + "completions/mean_length": 591.0, + "completions/min_length": 243.0, + "completions/max_length": 959.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 407.6000061035156, + "completions/min_terminated_length": 243.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4090277850627899, + "rewards/reward_func/std": 0.3618696630001068, + "reward": 0.4090277850627899, + "reward_std": 0.3618696331977844, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026517449878156185, + "sampling/sampling_logp_difference/max": 0.5514789819717407, + "sampling/importance_sampling_ratio/min": 0.27991440892219543, + "sampling/importance_sampling_ratio/mean": 0.905064046382904, + "sampling/importance_sampling_ratio/max": 1.3808459043502808, + "kl": 0.007381543560768478, + "entropy": 0.05520974611863494, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.289944043383002, + "epoch": 0.00666015625, + "step": 341 + }, + { + "loss": 0.16961312294006348, + "grad_norm": 2.0363802909851074, + "learning_rate": 1.5128205128205127e-07, + "num_tokens": 3347101.0, + "completions/mean_length": 602.625, + "completions/min_length": 226.0, + "completions/max_length": 926.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 437.20001220703125, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 926.0, + "tools/call_frequency": 14.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.46406251192092896, + "rewards/reward_func/std": 0.323446661233902, + "reward": 0.46406251192092896, + "reward_std": 0.323446661233902, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0023065595887601376, + "sampling/sampling_logp_difference/max": 0.354145884513855, + "sampling/importance_sampling_ratio/min": 0.6442764401435852, + "sampling/importance_sampling_ratio/mean": 1.0083256959915161, + "sampling/importance_sampling_ratio/max": 1.8325124979019165, + "kl": 0.006691869886708446, + "entropy": 0.06297752048703842, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.499040914699435, + "epoch": 0.0066796875, + "step": 342 + }, + { + "loss": 0.06328210234642029, + "grad_norm": 2.546358346939087, + "learning_rate": 1.4871794871794872e-07, + "num_tokens": 3356577.0, + "completions/mean_length": 500.5, + "completions/min_length": 141.0, + "completions/max_length": 911.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 259.20001220703125, + "completions/min_terminated_length": 141.0, + "completions/max_terminated_length": 349.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4613839387893677, + "rewards/reward_func/std": 0.4001639187335968, + "reward": 0.4613839387893677, + "reward_std": 0.4001638889312744, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0035084362607449293, + "sampling/sampling_logp_difference/max": 0.5969624519348145, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6001971960067749, + "sampling/importance_sampling_ratio/max": 0.9654102325439453, + "kl": 0.015198272834823001, + "entropy": 0.07368082262109965, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.728698687627912, + "epoch": 0.00669921875, + "step": 343 + }, + { + "loss": 0.21926726400852203, + "grad_norm": 1.4901846647262573, + "learning_rate": 1.4615384615384617e-07, + "num_tokens": 3367422.0, + "completions/mean_length": 671.0, + "completions/min_length": 283.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 531.4000244140625, + "completions/min_terminated_length": 283.0, + "completions/max_terminated_length": 918.0, + "tools/call_frequency": 16.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.32256942987442017, + "rewards/reward_func/std": 0.25267285108566284, + "reward": 0.32256942987442017, + "reward_std": 0.25267282128334045, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020809825509786606, + "sampling/sampling_logp_difference/max": 0.4671652317047119, + "sampling/importance_sampling_ratio/min": 0.5458290576934814, + "sampling/importance_sampling_ratio/mean": 0.8741626739501953, + "sampling/importance_sampling_ratio/max": 1.178560733795166, + "kl": 0.006418791912437882, + "entropy": 0.04946569993626326, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.052598850801587, + "epoch": 0.00671875, + "step": 344 + }, + { + "loss": 0.6059426069259644, + "grad_norm": 3.4927256107330322, + "learning_rate": 1.4358974358974356e-07, + "num_tokens": 3376337.0, + "completions/mean_length": 429.75, + "completions/min_length": 216.0, + "completions/max_length": 881.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 365.2857360839844, + "completions/min_terminated_length": 216.0, + "completions/max_terminated_length": 876.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.564707338809967, + "rewards/reward_func/std": 0.23219171166419983, + "reward": 0.564707338809967, + "reward_std": 0.23219171166419983, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00460396520793438, + "sampling/sampling_logp_difference/max": 0.4822502136230469, + "sampling/importance_sampling_ratio/min": 0.4994254410266876, + "sampling/importance_sampling_ratio/mean": 1.0679895877838135, + "sampling/importance_sampling_ratio/max": 2.2091777324676514, + "kl": 0.008540036338672508, + "entropy": 0.07662779104430228, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.341406114399433, + "epoch": 0.00673828125, + "step": 345 + }, + { + "loss": 1.12380850315094, + "grad_norm": 4.881765365600586, + "learning_rate": 1.4102564102564101e-07, + "num_tokens": 3385957.0, + "completions/mean_length": 517.625, + "completions/min_length": 235.0, + "completions/max_length": 984.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 265.20001220703125, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 307.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5052556991577148, + "rewards/reward_func/std": 0.30597010254859924, + "reward": 0.5052556991577148, + "reward_std": 0.30597007274627686, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003280271077528596, + "sampling/sampling_logp_difference/max": 1.0210515260696411, + "sampling/importance_sampling_ratio/min": 0.4611522853374481, + "sampling/importance_sampling_ratio/mean": 1.2063840627670288, + "sampling/importance_sampling_ratio/max": 2.183323860168457, + "kl": 0.007563284860225394, + "entropy": 0.06485960082500242, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.045484127476811, + "epoch": 0.0067578125, + "step": 346 + }, + { + "loss": 0.23536261916160583, + "grad_norm": 3.1003406047821045, + "learning_rate": 1.3846153846153846e-07, + "num_tokens": 3394809.0, + "completions/mean_length": 421.5, + "completions/min_length": 214.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 421.5, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 941.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47968751192092896, + "rewards/reward_func/std": 0.3380429148674011, + "reward": 0.47968751192092896, + "reward_std": 0.3380429148674011, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0031462633050978184, + "sampling/sampling_logp_difference/max": 0.6247248649597168, + "sampling/importance_sampling_ratio/min": 0.19461043179035187, + "sampling/importance_sampling_ratio/mean": 0.8709991574287415, + "sampling/importance_sampling_ratio/max": 1.2613483667373657, + "kl": 0.018713510638917796, + "entropy": 0.061064324487233534, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.221620719879866, + "epoch": 0.00677734375, + "step": 347 + }, + { + "loss": 0.3147105574607849, + "grad_norm": 2.1736278533935547, + "learning_rate": 1.3589743589743589e-07, + "num_tokens": 3404298.0, + "completions/mean_length": 501.5, + "completions/min_length": 213.0, + "completions/max_length": 906.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 367.16668701171875, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 897.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4694444537162781, + "rewards/reward_func/std": 0.31837037205696106, + "reward": 0.4694444537162781, + "reward_std": 0.31837034225463867, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004092707764357328, + "sampling/sampling_logp_difference/max": 0.443729043006897, + "sampling/importance_sampling_ratio/min": 0.3845905363559723, + "sampling/importance_sampling_ratio/mean": 0.713368833065033, + "sampling/importance_sampling_ratio/max": 1.083519458770752, + "kl": 0.005904047604417428, + "entropy": 0.07417159539181739, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.744835332036018, + "epoch": 0.006796875, + "step": 348 + }, + { + "loss": 0.22945883870124817, + "grad_norm": 1.811152696609497, + "learning_rate": 1.3333333333333334e-07, + "num_tokens": 3413568.0, + "completions/mean_length": 474.0, + "completions/min_length": 176.0, + "completions/max_length": 918.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 328.3333435058594, + "completions/min_terminated_length": 176.0, + "completions/max_terminated_length": 889.0, + "tools/call_frequency": 11.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33906251192092896, + "rewards/reward_func/std": 0.3942602574825287, + "reward": 0.33906251192092896, + "reward_std": 0.3942602872848511, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030528174247592688, + "sampling/sampling_logp_difference/max": 0.5464200973510742, + "sampling/importance_sampling_ratio/min": 0.15299411118030548, + "sampling/importance_sampling_ratio/mean": 0.8365310430526733, + "sampling/importance_sampling_ratio/max": 1.5227214097976685, + "kl": 0.014875005348585546, + "entropy": 0.0522242218721658, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.233148019760847, + "epoch": 0.00681640625, + "step": 349 + }, + { + "loss": 0.37245070934295654, + "grad_norm": 2.1516599655151367, + "learning_rate": 1.3076923076923076e-07, + "num_tokens": 3423250.0, + "completions/mean_length": 523.875, + "completions/min_length": 272.0, + "completions/max_length": 919.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 294.3999938964844, + "completions/min_terminated_length": 272.0, + "completions/max_terminated_length": 322.0, + "tools/call_frequency": 11.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.557329535484314, + "rewards/reward_func/std": 0.11384089291095734, + "reward": 0.557329535484314, + "reward_std": 0.11384088546037674, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004350366536527872, + "sampling/sampling_logp_difference/max": 0.7301149368286133, + "sampling/importance_sampling_ratio/min": 0.2895159125328064, + "sampling/importance_sampling_ratio/mean": 1.055694341659546, + "sampling/importance_sampling_ratio/max": 2.892615795135498, + "kl": 0.007862470272812061, + "entropy": 0.07687026262283325, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.958010666072369, + "epoch": 0.0068359375, + "step": 350 + }, + { + "loss": 0.1649133414030075, + "grad_norm": 2.9201619625091553, + "learning_rate": 1.2820512820512818e-07, + "num_tokens": 3432677.0, + "completions/mean_length": 495.0, + "completions/min_length": 233.0, + "completions/max_length": 906.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 363.66668701171875, + "completions/min_terminated_length": 233.0, + "completions/max_terminated_length": 847.0, + "tools/call_frequency": 12.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6052665114402771, + "rewards/reward_func/std": 0.29651540517807007, + "reward": 0.6052665114402771, + "reward_std": 0.29651540517807007, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030157535802572966, + "sampling/sampling_logp_difference/max": 0.6759873628616333, + "sampling/importance_sampling_ratio/min": 0.2782561779022217, + "sampling/importance_sampling_ratio/mean": 0.8316165208816528, + "sampling/importance_sampling_ratio/max": 1.4283554553985596, + "kl": 0.0075602816650643945, + "entropy": 0.06924465874908492, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.333971815183759, + "epoch": 0.00685546875, + "step": 351 + }, + { + "loss": -0.29376792907714844, + "grad_norm": 2.7973697185516357, + "learning_rate": 1.2564102564102563e-07, + "num_tokens": 3440833.0, + "completions/mean_length": 334.25, + "completions/min_length": 184.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 256.2857360839844, + "completions/min_terminated_length": 184.0, + "completions/max_terminated_length": 300.0, + "tools/call_frequency": 8.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6871875524520874, + "rewards/reward_func/std": 0.1200561597943306, + "reward": 0.6871875524520874, + "reward_std": 0.1200561448931694, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006464402191340923, + "sampling/sampling_logp_difference/max": 0.408966064453125, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6832584142684937, + "sampling/importance_sampling_ratio/max": 1.1695421934127808, + "kl": 0.014813631365541369, + "entropy": 0.10234812647104263, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.944047410041094, + "epoch": 0.006875, + "step": 352 + }, + { + "loss": 0.34225183725357056, + "grad_norm": 3.8913025856018066, + "learning_rate": 1.2307692307692308e-07, + "num_tokens": 3450517.0, + "completions/mean_length": 525.125, + "completions/min_length": 288.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 468.5714416503906, + "completions/min_terminated_length": 288.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 12.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4596022367477417, + "rewards/reward_func/std": 0.2831212282180786, + "reward": 0.4596022367477417, + "reward_std": 0.2831212282180786, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003586733480915427, + "sampling/sampling_logp_difference/max": 0.7742986679077148, + "sampling/importance_sampling_ratio/min": 0.8194791078567505, + "sampling/importance_sampling_ratio/mean": 1.4119439125061035, + "sampling/importance_sampling_ratio/max": 2.554555892944336, + "kl": 0.00603603120543994, + "entropy": 0.07299063808750361, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.124924259260297, + "epoch": 0.00689453125, + "step": 353 + }, + { + "loss": -0.04277166724205017, + "grad_norm": 2.2873141765594482, + "learning_rate": 1.205128205128205e-07, + "num_tokens": 3460603.0, + "completions/mean_length": 575.875, + "completions/min_length": 235.0, + "completions/max_length": 918.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 462.66668701171875, + "completions/min_terminated_length": 235.0, + "completions/max_terminated_length": 888.0, + "tools/call_frequency": 14.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4155145287513733, + "rewards/reward_func/std": 0.35393810272216797, + "reward": 0.4155145287513733, + "reward_std": 0.3539380729198456, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018601141637191176, + "sampling/sampling_logp_difference/max": 0.37059926986694336, + "sampling/importance_sampling_ratio/min": 0.5718969702720642, + "sampling/importance_sampling_ratio/mean": 0.914000391960144, + "sampling/importance_sampling_ratio/max": 1.253727912902832, + "kl": 0.006987865359405987, + "entropy": 0.04211903392570093, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.126300565898418, + "epoch": 0.0069140625, + "step": 354 + }, + { + "loss": 0.22429609298706055, + "grad_norm": 9.660198211669922, + "learning_rate": 1.1794871794871794e-07, + "num_tokens": 3471566.0, + "completions/mean_length": 682.875, + "completions/min_length": 226.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 533.0, + "completions/min_terminated_length": 226.0, + "completions/max_terminated_length": 932.0, + "tools/call_frequency": 15.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.24635416269302368, + "rewards/reward_func/std": 0.2638188600540161, + "reward": 0.24635416269302368, + "reward_std": 0.2638188600540161, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024306816048920155, + "sampling/sampling_logp_difference/max": 0.9216582775115967, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6891136765480042, + "sampling/importance_sampling_ratio/max": 1.1030634641647339, + "kl": 0.007123677482013591, + "entropy": 0.04444984998553991, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 14.541795387864113, + "epoch": 0.00693359375, + "step": 355 + }, + { + "loss": -0.11445406079292297, + "grad_norm": 1.6375746726989746, + "learning_rate": 1.1538461538461539e-07, + "num_tokens": 3481778.0, + "completions/mean_length": 591.75, + "completions/min_length": 214.0, + "completions/max_length": 914.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 405.3999938964844, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 914.0, + "tools/call_frequency": 14.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.2760416865348816, + "rewards/reward_func/std": 0.29846271872520447, + "reward": 0.2760416865348816, + "reward_std": 0.29846271872520447, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024060974828898907, + "sampling/sampling_logp_difference/max": 0.81072998046875, + "sampling/importance_sampling_ratio/min": 0.3293643295764923, + "sampling/importance_sampling_ratio/mean": 0.8496442437171936, + "sampling/importance_sampling_ratio/max": 1.1959024667739868, + "kl": 0.006176254144520499, + "entropy": 0.05185806841473095, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.438776256516576, + "epoch": 0.006953125, + "step": 356 + }, + { + "loss": 0.34434452652931213, + "grad_norm": 2.27691912651062, + "learning_rate": 1.1282051282051281e-07, + "num_tokens": 3492005.0, + "completions/mean_length": 592.75, + "completions/min_length": 227.0, + "completions/max_length": 925.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 485.0, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 905.0, + "tools/call_frequency": 14.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4071180820465088, + "rewards/reward_func/std": 0.31413763761520386, + "reward": 0.4071180820465088, + "reward_std": 0.31413763761520386, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002433410845696926, + "sampling/sampling_logp_difference/max": 0.4399832487106323, + "sampling/importance_sampling_ratio/min": 0.3618788719177246, + "sampling/importance_sampling_ratio/mean": 0.8325060606002808, + "sampling/importance_sampling_ratio/max": 1.4623525142669678, + "kl": 0.008023785252589732, + "entropy": 0.05365833907853812, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.345635615289211, + "epoch": 0.00697265625, + "step": 357 + }, + { + "loss": 0.09910199046134949, + "grad_norm": 5.544854640960693, + "learning_rate": 1.1025641025641025e-07, + "num_tokens": 3501704.0, + "completions/mean_length": 526.75, + "completions/min_length": 243.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 388.0, + "completions/min_terminated_length": 243.0, + "completions/max_terminated_length": 872.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.48520201444625854, + "rewards/reward_func/std": 0.22936289012432098, + "reward": 0.48520201444625854, + "reward_std": 0.2293628752231598, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0034862549509853125, + "sampling/sampling_logp_difference/max": 0.5000020861625671, + "sampling/importance_sampling_ratio/min": 0.2642839252948761, + "sampling/importance_sampling_ratio/mean": 0.9160890579223633, + "sampling/importance_sampling_ratio/max": 2.425163507461548, + "kl": 0.00727390666725114, + "entropy": 0.0725978160626255, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.401700576767325, + "epoch": 0.0069921875, + "step": 358 + }, + { + "loss": -0.076759934425354, + "grad_norm": 1.240573525428772, + "learning_rate": 1.076923076923077e-07, + "num_tokens": 3512669.0, + "completions/mean_length": 685.625, + "completions/min_length": 267.0, + "completions/max_length": 918.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 478.25, + "completions/min_terminated_length": 267.0, + "completions/max_terminated_length": 918.0, + "tools/call_frequency": 16.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4904018044471741, + "rewards/reward_func/std": 0.3291016221046448, + "reward": 0.4904018044471741, + "reward_std": 0.3291015923023224, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0018319201190024614, + "sampling/sampling_logp_difference/max": 0.38618898391723633, + "sampling/importance_sampling_ratio/min": 0.36532047390937805, + "sampling/importance_sampling_ratio/mean": 1.2976057529449463, + "sampling/importance_sampling_ratio/max": 2.8933873176574707, + "kl": 0.003742693952517584, + "entropy": 0.04015312288538553, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.82065217755735, + "epoch": 0.00701171875, + "step": 359 + }, + { + "loss": 0.23619353771209717, + "grad_norm": 2.01484751701355, + "learning_rate": 1.0512820512820512e-07, + "num_tokens": 3522859.0, + "completions/mean_length": 587.875, + "completions/min_length": 149.0, + "completions/max_length": 984.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 393.6000061035156, + "completions/min_terminated_length": 149.0, + "completions/max_terminated_length": 984.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.33715277910232544, + "rewards/reward_func/std": 0.3785034418106079, + "reward": 0.33715277910232544, + "reward_std": 0.3785034418106079, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0032063585240393877, + "sampling/sampling_logp_difference/max": 1.3778189420700073, + "sampling/importance_sampling_ratio/min": 0.325458824634552, + "sampling/importance_sampling_ratio/mean": 1.0155404806137085, + "sampling/importance_sampling_ratio/max": 2.142892360687256, + "kl": 0.011184677274286514, + "entropy": 0.05799563505570404, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.012726176530123, + "epoch": 0.00703125, + "step": 360 + }, + { + "loss": 0.20323601365089417, + "grad_norm": 1.9189603328704834, + "learning_rate": 1.0256410256410256e-07, + "num_tokens": 3533746.0, + "completions/mean_length": 676.375, + "completions/min_length": 221.0, + "completions/max_length": 927.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 436.75, + "completions/min_terminated_length": 221.0, + "completions/max_terminated_length": 903.0, + "tools/call_frequency": 15.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3947916626930237, + "rewards/reward_func/std": 0.3720635771751404, + "reward": 0.3947916626930237, + "reward_std": 0.3720635771751404, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024897779803723097, + "sampling/sampling_logp_difference/max": 0.717789888381958, + "sampling/importance_sampling_ratio/min": 0.679107666015625, + "sampling/importance_sampling_ratio/mean": 1.22198486328125, + "sampling/importance_sampling_ratio/max": 2.559250831604004, + "kl": 0.0035648090051836334, + "entropy": 0.0481795773957856, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.102604798972607, + "epoch": 0.00705078125, + "step": 361 + }, + { + "loss": 0.45477059483528137, + "grad_norm": 3.491288423538208, + "learning_rate": 1e-07, + "num_tokens": 3543308.0, + "completions/mean_length": 510.875, + "completions/min_length": 219.0, + "completions/max_length": 946.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 380.16668701171875, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 926.0, + "tools/call_frequency": 12.125, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35208332538604736, + "rewards/reward_func/std": 0.3326481282711029, + "reward": 0.35208332538604736, + "reward_std": 0.3326480984687805, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004156450740993023, + "sampling/sampling_logp_difference/max": 0.8912382125854492, + "sampling/importance_sampling_ratio/min": 0.23364315927028656, + "sampling/importance_sampling_ratio/mean": 0.7838151454925537, + "sampling/importance_sampling_ratio/max": 1.4487707614898682, + "kl": 0.007592372145154513, + "entropy": 0.07837582658976316, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.409661065787077, + "epoch": 0.0070703125, + "step": 362 + }, + { + "loss": 0.553205668926239, + "grad_norm": 2.654953718185425, + "learning_rate": 9.743589743589743e-08, + "num_tokens": 3554072.0, + "completions/mean_length": 661.25, + "completions/min_length": 200.0, + "completions/max_length": 986.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 233.0, + "completions/min_terminated_length": 200.0, + "completions/max_terminated_length": 275.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3239583373069763, + "rewards/reward_func/std": 0.3822821378707886, + "reward": 0.3239583373069763, + "reward_std": 0.3822821378707886, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017306222580373287, + "sampling/sampling_logp_difference/max": 1.2273404598236084, + "sampling/importance_sampling_ratio/min": 0.2814374268054962, + "sampling/importance_sampling_ratio/mean": 0.9925249814987183, + "sampling/importance_sampling_ratio/max": 2.144106388092041, + "kl": 0.006589077049284242, + "entropy": 0.038260414148680866, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.360424410551786, + "epoch": 0.00708984375, + "step": 363 + }, + { + "loss": 0.3046928644180298, + "grad_norm": 3.072150945663452, + "learning_rate": 9.487179487179487e-08, + "num_tokens": 3563231.0, + "completions/mean_length": 460.125, + "completions/min_length": 246.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 391.4285888671875, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 903.0, + "tools/call_frequency": 10.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.47547125816345215, + "rewards/reward_func/std": 0.24210003018379211, + "reward": 0.47547125816345215, + "reward_std": 0.24210000038146973, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004511426668614149, + "sampling/sampling_logp_difference/max": 0.4518766403198242, + "sampling/importance_sampling_ratio/min": 0.3715245723724365, + "sampling/importance_sampling_ratio/mean": 0.8560266494750977, + "sampling/importance_sampling_ratio/max": 1.4779739379882812, + "kl": 0.004891994409263134, + "entropy": 0.08420902397483587, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.915243545547128, + "epoch": 0.007109375, + "step": 364 + }, + { + "loss": -0.22046393156051636, + "grad_norm": 1.9495484828948975, + "learning_rate": 9.23076923076923e-08, + "num_tokens": 3573520.0, + "completions/mean_length": 600.375, + "completions/min_length": 277.0, + "completions/max_length": 915.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 420.8000183105469, + "completions/min_terminated_length": 277.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 14.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.41440972685813904, + "rewards/reward_func/std": 0.29290735721588135, + "reward": 0.41440972685813904, + "reward_std": 0.29290732741355896, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0022772413212805986, + "sampling/sampling_logp_difference/max": 0.42050549387931824, + "sampling/importance_sampling_ratio/min": 0.41630351543426514, + "sampling/importance_sampling_ratio/mean": 0.9943849444389343, + "sampling/importance_sampling_ratio/max": 1.393201470375061, + "kl": 0.006179814779898152, + "entropy": 0.05531756137497723, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.597842786461115, + "epoch": 0.00712890625, + "step": 365 + }, + { + "loss": 0.17228944599628448, + "grad_norm": 3.0747146606445312, + "learning_rate": 8.974358974358974e-08, + "num_tokens": 3584402.0, + "completions/mean_length": 675.625, + "completions/min_length": 196.0, + "completions/max_length": 956.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 432.0, + "completions/min_terminated_length": 196.0, + "completions/max_terminated_length": 897.0, + "tools/call_frequency": 15.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3161458373069763, + "rewards/reward_func/std": 0.32679933309555054, + "reward": 0.3161458373069763, + "reward_std": 0.32679933309555054, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019505694508552551, + "sampling/sampling_logp_difference/max": 0.7344026565551758, + "sampling/importance_sampling_ratio/min": 0.5828495621681213, + "sampling/importance_sampling_ratio/mean": 0.9005647301673889, + "sampling/importance_sampling_ratio/max": 1.3150254487991333, + "kl": 0.005911378211749252, + "entropy": 0.04034555109683424, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.827281272038817, + "epoch": 0.0071484375, + "step": 366 + }, + { + "loss": 0.04357209801673889, + "grad_norm": 1.601940631866455, + "learning_rate": 8.717948717948718e-08, + "num_tokens": 3594477.0, + "completions/mean_length": 574.125, + "completions/min_length": 220.0, + "completions/max_length": 940.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 257.75, + "completions/min_terminated_length": 220.0, + "completions/max_terminated_length": 334.0, + "tools/call_frequency": 13.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4394444525241852, + "rewards/reward_func/std": 0.31342506408691406, + "reward": 0.4394444525241852, + "reward_std": 0.31342506408691406, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0024494347162544727, + "sampling/sampling_logp_difference/max": 0.6245040893554688, + "sampling/importance_sampling_ratio/min": 0.48157235980033875, + "sampling/importance_sampling_ratio/mean": 0.7497169971466064, + "sampling/importance_sampling_ratio/max": 0.9611772894859314, + "kl": 0.005977083201287314, + "entropy": 0.054920990427490324, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.602703757584095, + "epoch": 0.00716796875, + "step": 367 + }, + { + "loss": 0.481326699256897, + "grad_norm": 3.005553722381592, + "learning_rate": 8.461538461538461e-08, + "num_tokens": 3604009.0, + "completions/mean_length": 506.25, + "completions/min_length": 205.0, + "completions/max_length": 932.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 445.4285888671875, + "completions/min_terminated_length": 205.0, + "completions/max_terminated_length": 926.0, + "tools/call_frequency": 12.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.43427082896232605, + "rewards/reward_func/std": 0.3862711787223816, + "reward": 0.43427082896232605, + "reward_std": 0.386271208524704, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002847282914444804, + "sampling/sampling_logp_difference/max": 0.4503697156906128, + "sampling/importance_sampling_ratio/min": 0.4040580689907074, + "sampling/importance_sampling_ratio/mean": 1.1304444074630737, + "sampling/importance_sampling_ratio/max": 2.4420998096466064, + "kl": 0.006565709292772226, + "entropy": 0.06570656324038282, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.262236099690199, + "epoch": 0.0071875, + "step": 368 + }, + { + "loss": 0.3552492558956146, + "grad_norm": 1.502150297164917, + "learning_rate": 8.205128205128205e-08, + "num_tokens": 3615450.0, + "completions/mean_length": 744.875, + "completions/min_length": 222.0, + "completions/max_length": 943.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 641.0, + "completions/min_terminated_length": 222.0, + "completions/max_terminated_length": 943.0, + "tools/call_frequency": 16.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.40565478801727295, + "rewards/reward_func/std": 0.37494614720344543, + "reward": 0.40565478801727295, + "reward_std": 0.3749461770057678, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.001546536455862224, + "sampling/sampling_logp_difference/max": 0.6686649322509766, + "sampling/importance_sampling_ratio/min": 0.32723268866539, + "sampling/importance_sampling_ratio/mean": 1.1739277839660645, + "sampling/importance_sampling_ratio/max": 2.3345067501068115, + "kl": 0.004890372933004983, + "entropy": 0.03535797132644802, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.44665022008121, + "epoch": 0.00720703125, + "step": 369 + }, + { + "loss": 0.4609681963920593, + "grad_norm": 2.5136477947235107, + "learning_rate": 7.948717948717947e-08, + "num_tokens": 3626257.0, + "completions/mean_length": 666.5, + "completions/min_length": 245.0, + "completions/max_length": 950.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 523.0, + "completions/min_terminated_length": 245.0, + "completions/max_terminated_length": 918.0, + "tools/call_frequency": 15.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4494791626930237, + "rewards/reward_func/std": 0.3661281168460846, + "reward": 0.4494791626930237, + "reward_std": 0.3661281168460846, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00303437071852386, + "sampling/sampling_logp_difference/max": 0.8716211318969727, + "sampling/importance_sampling_ratio/min": 0.536729633808136, + "sampling/importance_sampling_ratio/mean": 1.0293869972229004, + "sampling/importance_sampling_ratio/max": 2.0147554874420166, + "kl": 0.005898923933273181, + "entropy": 0.049022498656995595, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.27536160685122, + "epoch": 0.0072265625, + "step": 370 + }, + { + "loss": 0.3171140253543854, + "grad_norm": 1.8849818706512451, + "learning_rate": 7.692307692307692e-08, + "num_tokens": 3635703.0, + "completions/mean_length": 495.5, + "completions/min_length": 144.0, + "completions/max_length": 945.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 352.66668701171875, + "completions/min_terminated_length": 144.0, + "completions/max_terminated_length": 945.0, + "tools/call_frequency": 11.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.35572919249534607, + "rewards/reward_func/std": 0.30972927808761597, + "reward": 0.35572919249534607, + "reward_std": 0.30972927808761597, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002955238102003932, + "sampling/sampling_logp_difference/max": 0.4971802234649658, + "sampling/importance_sampling_ratio/min": 0.28106892108917236, + "sampling/importance_sampling_ratio/mean": 0.6937701106071472, + "sampling/importance_sampling_ratio/max": 1.32307767868042, + "kl": 0.008791801679763012, + "entropy": 0.062382240197621286, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.793375384062529, + "epoch": 0.00724609375, + "step": 371 + }, + { + "loss": 0.7845022678375244, + "grad_norm": 3.1953999996185303, + "learning_rate": 7.435897435897436e-08, + "num_tokens": 3645262.0, + "completions/mean_length": 510.625, + "completions/min_length": 229.0, + "completions/max_length": 963.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 255.0, + "completions/min_terminated_length": 229.0, + "completions/max_terminated_length": 286.0, + "tools/call_frequency": 11.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5342013835906982, + "rewards/reward_func/std": 0.3883250951766968, + "reward": 0.5342013835906982, + "reward_std": 0.3883250951766968, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.003755297977477312, + "sampling/sampling_logp_difference/max": 0.41947174072265625, + "sampling/importance_sampling_ratio/min": 0.3582685887813568, + "sampling/importance_sampling_ratio/mean": 1.4003347158432007, + "sampling/importance_sampling_ratio/max": 2.905799150466919, + "kl": 0.007793017070071073, + "entropy": 0.0703104060376063, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.175190338864923, + "epoch": 0.007265625, + "step": 372 + }, + { + "loss": -0.09775468707084656, + "grad_norm": 1.4430351257324219, + "learning_rate": 7.179487179487178e-08, + "num_tokens": 3656071.0, + "completions/mean_length": 665.125, + "completions/min_length": 251.0, + "completions/max_length": 911.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 522.7999877929688, + "completions/min_terminated_length": 251.0, + "completions/max_terminated_length": 905.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.42239582538604736, + "rewards/reward_func/std": 0.3452551066875458, + "reward": 0.42239582538604736, + "reward_std": 0.3452551066875458, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0014720268081873655, + "sampling/sampling_logp_difference/max": 0.3808107376098633, + "sampling/importance_sampling_ratio/min": 0.47723719477653503, + "sampling/importance_sampling_ratio/mean": 0.9883498549461365, + "sampling/importance_sampling_ratio/max": 1.7035512924194336, + "kl": 0.005829234898556024, + "entropy": 0.03751454464509152, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.049248242750764, + "epoch": 0.00728515625, + "step": 373 + }, + { + "loss": -0.13575752079486847, + "grad_norm": 3.0597567558288574, + "learning_rate": 6.923076923076923e-08, + "num_tokens": 3665546.0, + "completions/mean_length": 498.375, + "completions/min_length": 218.0, + "completions/max_length": 880.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 443.857177734375, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 845.0, + "tools/call_frequency": 12.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5932133793830872, + "rewards/reward_func/std": 0.22024747729301453, + "reward": 0.5932133793830872, + "reward_std": 0.22024746239185333, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0031017200089991093, + "sampling/sampling_logp_difference/max": 0.48731911182403564, + "sampling/importance_sampling_ratio/min": 0.43211644887924194, + "sampling/importance_sampling_ratio/mean": 0.9861002564430237, + "sampling/importance_sampling_ratio/max": 1.6476496458053589, + "kl": 0.0067885682510677725, + "entropy": 0.06060576264280826, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.637401569634676, + "epoch": 0.0073046875, + "step": 374 + }, + { + "loss": 0.05237816274166107, + "grad_norm": 1.3126393556594849, + "learning_rate": 6.666666666666667e-08, + "num_tokens": 3676933.0, + "completions/mean_length": 738.875, + "completions/min_length": 246.0, + "completions/max_length": 920.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 590.75, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 894.0, + "tools/call_frequency": 17.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.34600692987442017, + "rewards/reward_func/std": 0.3261962831020355, + "reward": 0.34600692987442017, + "reward_std": 0.32619625329971313, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017476590583100915, + "sampling/sampling_logp_difference/max": 0.7030580043792725, + "sampling/importance_sampling_ratio/min": 0.8213844299316406, + "sampling/importance_sampling_ratio/mean": 1.0222153663635254, + "sampling/importance_sampling_ratio/max": 1.5689725875854492, + "kl": 0.005186290480196476, + "entropy": 0.03765083180041984, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.231354838237166, + "epoch": 0.00732421875, + "step": 375 + }, + { + "loss": 0.22939026355743408, + "grad_norm": 4.588799476623535, + "learning_rate": 6.410256410256409e-08, + "num_tokens": 3686441.0, + "completions/mean_length": 503.375, + "completions/min_length": 224.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 442.5714416503906, + "completions/min_terminated_length": 224.0, + "completions/max_terminated_length": 953.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4593749940395355, + "rewards/reward_func/std": 0.26946771144866943, + "reward": 0.4593749940395355, + "reward_std": 0.26946768164634705, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00392577238380909, + "sampling/sampling_logp_difference/max": 1.753282070159912, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.6618277430534363, + "sampling/importance_sampling_ratio/max": 1.3406578302383423, + "kl": 0.006150438901386224, + "entropy": 0.07508856005733833, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.451407862827182, + "epoch": 0.00734375, + "step": 376 + }, + { + "loss": 0.041976217180490494, + "grad_norm": 2.9687867164611816, + "learning_rate": 6.153846153846154e-08, + "num_tokens": 3697023.0, + "completions/mean_length": 637.0, + "completions/min_length": 183.0, + "completions/max_length": 930.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 398.75, + "completions/min_terminated_length": 183.0, + "completions/max_terminated_length": 916.0, + "tools/call_frequency": 15.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.37522321939468384, + "rewards/reward_func/std": 0.3589252531528473, + "reward": 0.37522321939468384, + "reward_std": 0.3589252233505249, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002270605880767107, + "sampling/sampling_logp_difference/max": 1.1105540990829468, + "sampling/importance_sampling_ratio/min": 0.23670749366283417, + "sampling/importance_sampling_ratio/mean": 0.9392521381378174, + "sampling/importance_sampling_ratio/max": 1.6178677082061768, + "kl": 0.0062582664540968835, + "entropy": 0.04221249872352928, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.812686471268535, + "epoch": 0.00736328125, + "step": 377 + }, + { + "loss": -0.21108978986740112, + "grad_norm": 1.4912251234054565, + "learning_rate": 5.897435897435897e-08, + "num_tokens": 3707785.0, + "completions/mean_length": 660.625, + "completions/min_length": 246.0, + "completions/max_length": 911.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 434.5, + "completions/min_terminated_length": 246.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 15.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5777380466461182, + "rewards/reward_func/std": 0.2766937017440796, + "reward": 0.5777380466461182, + "reward_std": 0.276693731546402, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002035868354141712, + "sampling/sampling_logp_difference/max": 0.6254613399505615, + "sampling/importance_sampling_ratio/min": 0.4704410433769226, + "sampling/importance_sampling_ratio/mean": 1.1512235403060913, + "sampling/importance_sampling_ratio/max": 1.539965271949768, + "kl": 0.006231733888853341, + "entropy": 0.04667031962890178, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.149119028821588, + "epoch": 0.0073828125, + "step": 378 + }, + { + "loss": 0.2814968228340149, + "grad_norm": 1.5624507665634155, + "learning_rate": 5.641025641025641e-08, + "num_tokens": 3718574.0, + "completions/mean_length": 663.375, + "completions/min_length": 225.0, + "completions/max_length": 929.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 249.6666717529297, + "completions/min_terminated_length": 225.0, + "completions/max_terminated_length": 292.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3177083134651184, + "rewards/reward_func/std": 0.3765493929386139, + "reward": 0.3177083134651184, + "reward_std": 0.3765493631362915, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0016697872197255492, + "sampling/sampling_logp_difference/max": 0.3796495199203491, + "sampling/importance_sampling_ratio/min": 0.44915464520454407, + "sampling/importance_sampling_ratio/mean": 0.7563238739967346, + "sampling/importance_sampling_ratio/max": 1.0380362272262573, + "kl": 0.00667185103520751, + "entropy": 0.041635910514742136, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.71958476677537, + "epoch": 0.00740234375, + "step": 379 + }, + { + "loss": 0.27018266916275024, + "grad_norm": 2.250265121459961, + "learning_rate": 5.384615384615385e-08, + "num_tokens": 3728215.0, + "completions/mean_length": 520.0, + "completions/min_length": 204.0, + "completions/max_length": 948.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 269.3999938964844, + "completions/min_terminated_length": 204.0, + "completions/max_terminated_length": 408.0, + "tools/call_frequency": 11.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3863541781902313, + "rewards/reward_func/std": 0.3658386170864105, + "reward": 0.3863541781902313, + "reward_std": 0.3658386170864105, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030760220251977444, + "sampling/sampling_logp_difference/max": 0.5609376430511475, + "sampling/importance_sampling_ratio/min": 0.6947159767150879, + "sampling/importance_sampling_ratio/mean": 0.939835786819458, + "sampling/importance_sampling_ratio/max": 1.3812968730926514, + "kl": 0.01511119284259621, + "entropy": 0.06744975305628031, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.221181690692902, + "epoch": 0.007421875, + "step": 380 + }, + { + "loss": 0.35851940512657166, + "grad_norm": 4.0134735107421875, + "learning_rate": 5.128205128205128e-08, + "num_tokens": 3737050.0, + "completions/mean_length": 419.375, + "completions/min_length": 211.0, + "completions/max_length": 941.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 257.66668701171875, + "completions/min_terminated_length": 211.0, + "completions/max_terminated_length": 318.0, + "tools/call_frequency": 10.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4701231122016907, + "rewards/reward_func/std": 0.3291320502758026, + "reward": 0.4701231122016907, + "reward_std": 0.3291320502758026, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004206265788525343, + "sampling/sampling_logp_difference/max": 0.4848172664642334, + "sampling/importance_sampling_ratio/min": 0.22882135212421417, + "sampling/importance_sampling_ratio/mean": 0.8985360264778137, + "sampling/importance_sampling_ratio/max": 2.0683910846710205, + "kl": 0.005411783655290492, + "entropy": 0.07788842858280987, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.53570157289505, + "epoch": 0.00744140625, + "step": 381 + }, + { + "loss": 0.0343230701982975, + "grad_norm": 1.2299612760543823, + "learning_rate": 4.8717948717948716e-08, + "num_tokens": 3748473.0, + "completions/mean_length": 742.25, + "completions/min_length": 270.0, + "completions/max_length": 943.0, + "completions/clipped_ratio": 0.625, + "completions/mean_terminated_length": 484.3333435058594, + "completions/min_terminated_length": 270.0, + "completions/max_terminated_length": 900.0, + "tools/call_frequency": 17.375, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5300978422164917, + "rewards/reward_func/std": 0.31030702590942383, + "reward": 0.5300978422164917, + "reward_std": 0.3103070557117462, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0017211101949214935, + "sampling/sampling_logp_difference/max": 0.44815826416015625, + "sampling/importance_sampling_ratio/min": 0.508392333984375, + "sampling/importance_sampling_ratio/mean": 0.8790225982666016, + "sampling/importance_sampling_ratio/max": 1.3396621942520142, + "kl": 0.0050737156852846965, + "entropy": 0.03387230832595378, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.965210551396012, + "epoch": 0.0074609375, + "step": 382 + }, + { + "loss": 0.2191891372203827, + "grad_norm": 2.4144206047058105, + "learning_rate": 4.615384615384615e-08, + "num_tokens": 3759078.0, + "completions/mean_length": 640.875, + "completions/min_length": 206.0, + "completions/max_length": 907.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 564.1666870117188, + "completions/min_terminated_length": 206.0, + "completions/max_terminated_length": 907.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.48608630895614624, + "rewards/reward_func/std": 0.3177451491355896, + "reward": 0.48608630895614624, + "reward_std": 0.317745178937912, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.00187301158439368, + "sampling/sampling_logp_difference/max": 0.4956684112548828, + "sampling/importance_sampling_ratio/min": 0.5989976525306702, + "sampling/importance_sampling_ratio/mean": 0.9288737773895264, + "sampling/importance_sampling_ratio/max": 1.2974257469177246, + "kl": 0.007222484840895049, + "entropy": 0.034451300860382617, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.321598336100578, + "epoch": 0.00748046875, + "step": 383 + }, + { + "loss": -0.22111737728118896, + "grad_norm": 1.7970424890518188, + "learning_rate": 4.358974358974359e-08, + "num_tokens": 3768682.0, + "completions/mean_length": 515.5, + "completions/min_length": 266.0, + "completions/max_length": 899.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 391.16668701171875, + "completions/min_terminated_length": 266.0, + "completions/max_terminated_length": 860.0, + "tools/call_frequency": 12.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.40551450848579407, + "rewards/reward_func/std": 0.23584631085395813, + "reward": 0.40551450848579407, + "reward_std": 0.23584629595279694, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0040244124829769135, + "sampling/sampling_logp_difference/max": 0.7685341835021973, + "sampling/importance_sampling_ratio/min": 0.0, + "sampling/importance_sampling_ratio/mean": 0.7111059427261353, + "sampling/importance_sampling_ratio/max": 1.2578823566436768, + "kl": 0.010656450263923034, + "entropy": 0.06825953623047099, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.836524432525039, + "epoch": 0.0075, + "step": 384 + }, + { + "loss": 0.32538095116615295, + "grad_norm": 2.290262222290039, + "learning_rate": 4.1025641025641025e-08, + "num_tokens": 3778738.0, + "completions/mean_length": 572.0, + "completions/min_length": 237.0, + "completions/max_length": 959.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 374.0, + "completions/min_terminated_length": 237.0, + "completions/max_terminated_length": 887.0, + "tools/call_frequency": 13.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5785353183746338, + "rewards/reward_func/std": 0.3126175105571747, + "reward": 0.5785353183746338, + "reward_std": 0.3126174807548523, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002567760180681944, + "sampling/sampling_logp_difference/max": 0.4436044692993164, + "sampling/importance_sampling_ratio/min": 0.5286172032356262, + "sampling/importance_sampling_ratio/mean": 1.0732473134994507, + "sampling/importance_sampling_ratio/max": 2.0339035987854004, + "kl": 0.004483790398808196, + "entropy": 0.0601368339266628, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.025060197338462, + "epoch": 0.00751953125, + "step": 385 + }, + { + "loss": 0.8102378845214844, + "grad_norm": 3.486389636993408, + "learning_rate": 3.846153846153846e-08, + "num_tokens": 3788879.0, + "completions/mean_length": 584.0, + "completions/min_length": 227.0, + "completions/max_length": 937.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 275.75, + "completions/min_terminated_length": 227.0, + "completions/max_terminated_length": 317.0, + "tools/call_frequency": 13.625, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6276041269302368, + "rewards/reward_func/std": 0.36699753999710083, + "reward": 0.6276041269302368, + "reward_std": 0.36699751019477844, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0029594649095088243, + "sampling/sampling_logp_difference/max": 0.5997998714447021, + "sampling/importance_sampling_ratio/min": 0.663475751876831, + "sampling/importance_sampling_ratio/mean": 0.9594264030456543, + "sampling/importance_sampling_ratio/max": 1.9654589891433716, + "kl": 0.005660253555106465, + "entropy": 0.06197942903963849, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.371305882930756, + "epoch": 0.0075390625, + "step": 386 + }, + { + "loss": 0.2293315827846527, + "grad_norm": 4.022679328918457, + "learning_rate": 3.589743589743589e-08, + "num_tokens": 3797904.0, + "completions/mean_length": 442.375, + "completions/min_length": 228.0, + "completions/max_length": 896.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 442.375, + "completions/min_terminated_length": 228.0, + "completions/max_terminated_length": 896.0, + "tools/call_frequency": 10.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5395833253860474, + "rewards/reward_func/std": 0.24779784679412842, + "reward": 0.5395833253860474, + "reward_std": 0.24779784679412842, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004252722952514887, + "sampling/sampling_logp_difference/max": 0.45195960998535156, + "sampling/importance_sampling_ratio/min": 0.5976433753967285, + "sampling/importance_sampling_ratio/mean": 1.1657466888427734, + "sampling/importance_sampling_ratio/max": 2.289912700653076, + "kl": 0.007372680003754795, + "entropy": 0.08194333827123046, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.542544173076749, + "epoch": 0.00755859375, + "step": 387 + }, + { + "loss": 0.552078366279602, + "grad_norm": 2.3305625915527344, + "learning_rate": 3.3333333333333334e-08, + "num_tokens": 3808015.0, + "completions/mean_length": 579.75, + "completions/min_length": 236.0, + "completions/max_length": 937.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 258.5, + "completions/min_terminated_length": 236.0, + "completions/max_terminated_length": 305.0, + "tools/call_frequency": 13.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.48385417461395264, + "rewards/reward_func/std": 0.3341020941734314, + "reward": 0.48385417461395264, + "reward_std": 0.334102064371109, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0022845868952572346, + "sampling/sampling_logp_difference/max": 0.5403456687927246, + "sampling/importance_sampling_ratio/min": 0.5095613598823547, + "sampling/importance_sampling_ratio/mean": 1.0987298488616943, + "sampling/importance_sampling_ratio/max": 1.8140168190002441, + "kl": 0.0060094244399806485, + "entropy": 0.04960021167062223, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.707540176808834, + "epoch": 0.007578125, + "step": 388 + }, + { + "loss": -0.07801030576229095, + "grad_norm": 1.4224897623062134, + "learning_rate": 3.076923076923077e-08, + "num_tokens": 3816543.0, + "completions/mean_length": 381.25, + "completions/min_length": 17.0, + "completions/max_length": 933.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 381.25, + "completions/min_terminated_length": 17.0, + "completions/max_terminated_length": 933.0, + "tools/call_frequency": 9.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5534672737121582, + "rewards/reward_func/std": 0.3856430947780609, + "reward": 0.5534672737121582, + "reward_std": 0.3856430947780609, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0033290949650108814, + "sampling/sampling_logp_difference/max": 0.7301068305969238, + "sampling/importance_sampling_ratio/min": 0.3669135570526123, + "sampling/importance_sampling_ratio/mean": 0.8244848847389221, + "sampling/importance_sampling_ratio/max": 1.3798127174377441, + "kl": 0.008750877423153725, + "entropy": 0.06230661051813513, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 10.940627517178655, + "epoch": 0.00759765625, + "step": 389 + }, + { + "loss": 0.3872295022010803, + "grad_norm": 1.4772762060165405, + "learning_rate": 2.8205128205128203e-08, + "num_tokens": 3826072.0, + "completions/mean_length": 506.5, + "completions/min_length": 234.0, + "completions/max_length": 952.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 267.20001220703125, + "completions/min_terminated_length": 234.0, + "completions/max_terminated_length": 313.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5309027433395386, + "rewards/reward_func/std": 0.2569393515586853, + "reward": 0.5309027433395386, + "reward_std": 0.2569393515586853, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0026092235930263996, + "sampling/sampling_logp_difference/max": 0.5579478740692139, + "sampling/importance_sampling_ratio/min": 0.5833961367607117, + "sampling/importance_sampling_ratio/mean": 0.8792951703071594, + "sampling/importance_sampling_ratio/max": 1.3834611177444458, + "kl": 0.005616139598714653, + "entropy": 0.060529900423716754, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.814008753746748, + "epoch": 0.0076171875, + "step": 390 + }, + { + "loss": 0.00789196789264679, + "grad_norm": 4.139995098114014, + "learning_rate": 2.564102564102564e-08, + "num_tokens": 3833571.0, + "completions/mean_length": 251.5, + "completions/min_length": 189.0, + "completions/max_length": 311.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 251.5, + "completions/min_terminated_length": 189.0, + "completions/max_terminated_length": 311.0, + "tools/call_frequency": 6.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.6579117178916931, + "rewards/reward_func/std": 0.16509607434272766, + "reward": 0.6579117178916931, + "reward_std": 0.16509607434272766, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.006336327642202377, + "sampling/sampling_logp_difference/max": 0.5054349899291992, + "sampling/importance_sampling_ratio/min": 0.351392924785614, + "sampling/importance_sampling_ratio/mean": 0.721153736114502, + "sampling/importance_sampling_ratio/max": 1.3322499990463257, + "kl": 0.010527876438573003, + "entropy": 0.08685639686882496, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 6.415014378726482, + "epoch": 0.00763671875, + "step": 391 + }, + { + "loss": 0.46771329641342163, + "grad_norm": 3.8041813373565674, + "learning_rate": 2.3076923076923076e-08, + "num_tokens": 3842531.0, + "completions/mean_length": 435.75, + "completions/min_length": 245.0, + "completions/max_length": 915.0, + "completions/clipped_ratio": 0.0, + "completions/mean_terminated_length": 435.75, + "completions/min_terminated_length": 245.0, + "completions/max_terminated_length": 915.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44739586114883423, + "rewards/reward_func/std": 0.32612699270248413, + "reward": 0.44739586114883423, + "reward_std": 0.3261270225048065, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0039434051141142845, + "sampling/sampling_logp_difference/max": 0.3881545066833496, + "sampling/importance_sampling_ratio/min": 0.6921693086624146, + "sampling/importance_sampling_ratio/mean": 1.196067214012146, + "sampling/importance_sampling_ratio/max": 1.9909387826919556, + "kl": 0.008468287764117122, + "entropy": 0.08524313545785844, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.187724016606808, + "epoch": 0.00765625, + "step": 392 + }, + { + "loss": 0.7142651081085205, + "grad_norm": 2.796743631362915, + "learning_rate": 2.0512820512820512e-08, + "num_tokens": 3851506.0, + "completions/mean_length": 438.0, + "completions/min_length": 218.0, + "completions/max_length": 921.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 369.0000305175781, + "completions/min_terminated_length": 218.0, + "completions/max_terminated_length": 898.0, + "tools/call_frequency": 10.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.43861111998558044, + "rewards/reward_func/std": 0.2780018150806427, + "reward": 0.43861111998558044, + "reward_std": 0.2780018150806427, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004804760683327913, + "sampling/sampling_logp_difference/max": 0.4846458435058594, + "sampling/importance_sampling_ratio/min": 0.3469865024089813, + "sampling/importance_sampling_ratio/mean": 0.9689599871635437, + "sampling/importance_sampling_ratio/max": 2.09224009513855, + "kl": 0.012090899239410646, + "entropy": 0.09989614365622401, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.768175946548581, + "epoch": 0.00767578125, + "step": 393 + }, + { + "loss": 0.3994956612586975, + "grad_norm": 1.435256838798523, + "learning_rate": 1.7948717948717946e-08, + "num_tokens": 3861525.0, + "completions/mean_length": 568.75, + "completions/min_length": 193.0, + "completions/max_length": 942.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 448.16668701171875, + "completions/min_terminated_length": 193.0, + "completions/max_terminated_length": 925.0, + "tools/call_frequency": 13.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3122064471244812, + "rewards/reward_func/std": 0.28681254386901855, + "reward": 0.3122064471244812, + "reward_std": 0.28681254386901855, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0025168871507048607, + "sampling/sampling_logp_difference/max": 0.9437516927719116, + "sampling/importance_sampling_ratio/min": 0.20858502388000488, + "sampling/importance_sampling_ratio/mean": 0.7111230492591858, + "sampling/importance_sampling_ratio/max": 1.397501826286316, + "kl": 0.010764957100036554, + "entropy": 0.05726829695049673, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.074693532660604, + "epoch": 0.0076953125, + "step": 394 + }, + { + "loss": -0.1842808574438095, + "grad_norm": 1.8956371545791626, + "learning_rate": 1.5384615384615385e-08, + "num_tokens": 3870958.0, + "completions/mean_length": 492.75, + "completions/min_length": 207.0, + "completions/max_length": 900.0, + "completions/clipped_ratio": 0.25, + "completions/mean_terminated_length": 357.5, + "completions/min_terminated_length": 207.0, + "completions/max_terminated_length": 891.0, + "tools/call_frequency": 11.75, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4130208194255829, + "rewards/reward_func/std": 0.3744204342365265, + "reward": 0.4130208194255829, + "reward_std": 0.3744204342365265, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0030384541023522615, + "sampling/sampling_logp_difference/max": 0.40828776359558105, + "sampling/importance_sampling_ratio/min": 0.36701667308807373, + "sampling/importance_sampling_ratio/mean": 0.7287545204162598, + "sampling/importance_sampling_ratio/max": 1.0789176225662231, + "kl": 0.012043392693158239, + "entropy": 0.06867672456428409, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.308229077607393, + "epoch": 0.00771484375, + "step": 395 + }, + { + "loss": 0.28571486473083496, + "grad_norm": 1.5805855989456177, + "learning_rate": 1.282051282051282e-08, + "num_tokens": 3881152.0, + "completions/mean_length": 587.25, + "completions/min_length": 251.0, + "completions/max_length": 928.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 411.3999938964844, + "completions/min_terminated_length": 251.0, + "completions/max_terminated_length": 905.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.44861114025115967, + "rewards/reward_func/std": 0.20515163242816925, + "reward": 0.44861114025115967, + "reward_std": 0.20515163242816925, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.002903548302128911, + "sampling/sampling_logp_difference/max": 0.40503907203674316, + "sampling/importance_sampling_ratio/min": 0.3894405961036682, + "sampling/importance_sampling_ratio/mean": 0.8523924350738525, + "sampling/importance_sampling_ratio/max": 1.6263105869293213, + "kl": 0.006303388392552733, + "entropy": 0.05782128660939634, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.868696028366685, + "epoch": 0.007734375, + "step": 396 + }, + { + "loss": 0.10849229991436005, + "grad_norm": 1.920297384262085, + "learning_rate": 1.0256410256410256e-08, + "num_tokens": 3891336.0, + "completions/mean_length": 587.125, + "completions/min_length": 219.0, + "completions/max_length": 953.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 391.20001220703125, + "completions/min_terminated_length": 219.0, + "completions/max_terminated_length": 953.0, + "tools/call_frequency": 13.5, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.5357639193534851, + "rewards/reward_func/std": 0.36145415902137756, + "reward": 0.5357639193534851, + "reward_std": 0.36145415902137756, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020170961506664753, + "sampling/sampling_logp_difference/max": 0.38814985752105713, + "sampling/importance_sampling_ratio/min": 0.4338534474372864, + "sampling/importance_sampling_ratio/mean": 0.8134323954582214, + "sampling/importance_sampling_ratio/max": 1.1221262216567993, + "kl": 0.007870797271607444, + "entropy": 0.049492064339574426, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.554226817563176, + "epoch": 0.00775390625, + "step": 397 + }, + { + "loss": 0.029680661857128143, + "grad_norm": 1.1214306354522705, + "learning_rate": 7.692307692307693e-09, + "num_tokens": 3902233.0, + "completions/mean_length": 676.625, + "completions/min_length": 273.0, + "completions/max_length": 954.0, + "completions/clipped_ratio": 0.5, + "completions/mean_terminated_length": 448.75, + "completions/min_terminated_length": 273.0, + "completions/max_terminated_length": 920.0, + "tools/call_frequency": 15.875, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.40937501192092896, + "rewards/reward_func/std": 0.31125345826148987, + "reward": 0.40937501192092896, + "reward_std": 0.31125345826148987, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0019542539957910776, + "sampling/sampling_logp_difference/max": 0.737324595451355, + "sampling/importance_sampling_ratio/min": 0.3489941656589508, + "sampling/importance_sampling_ratio/mean": 0.627144992351532, + "sampling/importance_sampling_ratio/max": 1.0400105714797974, + "kl": 0.00467789861431811, + "entropy": 0.04684537241701037, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 13.040527725592256, + "epoch": 0.0077734375, + "step": 398 + }, + { + "loss": 0.3116846978664398, + "grad_norm": 4.387675762176514, + "learning_rate": 5.128205128205128e-09, + "num_tokens": 3911129.0, + "completions/mean_length": 426.875, + "completions/min_length": 214.0, + "completions/max_length": 938.0, + "completions/clipped_ratio": 0.125, + "completions/mean_terminated_length": 353.8571472167969, + "completions/min_terminated_length": 214.0, + "completions/max_terminated_length": 813.0, + "tools/call_frequency": 10.25, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.3879103660583496, + "rewards/reward_func/std": 0.2869396209716797, + "reward": 0.3879103660583496, + "reward_std": 0.2869396209716797, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.004121157340705395, + "sampling/sampling_logp_difference/max": 0.43259263038635254, + "sampling/importance_sampling_ratio/min": 0.3432130217552185, + "sampling/importance_sampling_ratio/mean": 1.1767027378082275, + "sampling/importance_sampling_ratio/max": 2.0975093841552734, + "kl": 0.007573936520202551, + "entropy": 0.08144226239528507, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 11.414660146459937, + "epoch": 0.00779296875, + "step": 399 + }, + { + "loss": 0.006380332633852959, + "grad_norm": 1.5772380828857422, + "learning_rate": 2.564102564102564e-09, + "num_tokens": 3921134.0, + "completions/mean_length": 565.625, + "completions/min_length": 213.0, + "completions/max_length": 895.0, + "completions/clipped_ratio": 0.375, + "completions/mean_terminated_length": 387.6000061035156, + "completions/min_terminated_length": 213.0, + "completions/max_terminated_length": 886.0, + "tools/call_frequency": 15.0, + "tools/failure_frequency": 0.0, + "rewards/reward_func/mean": 0.4609375, + "rewards/reward_func/std": 0.27517977356910706, + "reward": 0.4609375, + "reward_std": 0.27517977356910706, + "frac_reward_zero_std": 0.0, + "sampling/sampling_logp_difference/mean": 0.0020603733137249947, + "sampling/sampling_logp_difference/max": 0.4640183448791504, + "sampling/importance_sampling_ratio/min": 0.6881456971168518, + "sampling/importance_sampling_ratio/mean": 1.0388755798339844, + "sampling/importance_sampling_ratio/max": 1.8433419466018677, + "kl": 0.00784167206438724, + "entropy": 0.052742726111318916, + "clip_ratio/low_mean": 0.0, + "clip_ratio/low_min": 0.0, + "clip_ratio/high_mean": 0.0, + "clip_ratio/high_max": 0.0, + "clip_ratio/region_mean": 0.0, + "step_time": 12.191913763061166, + "epoch": 0.0078125, + "step": 400 + }, + { + "train_runtime": 5646.2026, + "train_samples_per_second": 0.567, + "train_steps_per_second": 0.071, + "total_flos": 0.0, + "train_loss": 0.16429192219395192, + "epoch": 0.0078125, + "step": 400 + } +] \ No newline at end of file diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..4239c79 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:353035b86b57bf4382188f85de1825c9f04ad78c84c8eb0dc9253db627eae900 +size 6882335328 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..c7afbed --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 +size 11422650 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..af5f35b --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,75 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "local_files_only": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "padding_side": "left", + "response_schema": { + "properties": { + "content": { + "type": "string" + }, + "reasoning_content": { + "type": "string" + }, + "role": { + "const": "assistant" + }, + "tool_calls": { + "items": { + "properties": { + "function": { + "properties": { + "arguments": { + "additionalProperties": {}, + "type": "object" + }, + "name": { + "type": "string" + } + }, + "type": "object" + }, + "type": { + "const": "function" + } + }, + "type": "object", + "x-parser": "json", + "x-parser-args": { + "transform": "{type: 'function', function: @}" + } + }, + "type": "array", + "x-regex-iterator": "\\s*(.+?)\\s*" + } + }, + "type": "object", + "x-regex": "^(?:\\n?(?:(?P.*?\\S.*?)\\n?|[\\s]*)\\s*)?(?P.*?)(?:\\n(?=))?(?=(?:|<\\|im_end\\|>|$))(?P(?:.+?\\s*)+)?\\s*(?:<\\|im_end\\|>|$)" + }, + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "truncation_side": "left", + "unk_token": null +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000..cf67172 --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0bd88a3e09989a53fb2b010ff17340af774a6706c36624da53f72828201c98bc +size 7249 diff --git a/training_summary.json b/training_summary.json new file mode 100644 index 0000000..8d7734c --- /dev/null +++ b/training_summary.json @@ -0,0 +1,15 @@ +{ + "model": "Qwen/Qwen3-1.7B", + "max_steps": 400, + "num_generations": 8, + "vllm_gpu_memory_utilization": 0.55, + "max_completion_length": 1536, + "train_seconds": 5693.903043985367, + "stats": "TrainOutput(global_step=400, training_loss=0.16429192219395192, metrics={'train_runtime': 5646.2026, 'train_samples_per_second': 0.567, 'train_steps_per_second': 0.071, 'total_flos': 0.0, 'train_loss': 0.16429192219395192})", + "failed": false, + "failure_reason": "", + "output_dir": "clarify-rl-grpo-qwen3-1-7b-run7", + "trackio_space_id": "clarify-rl-grpo-qwen3-1-7b-run7", + "num_log_entries": 401, + "smoke_test": false +} \ No newline at end of file