初始化项目,由ModelHub XC社区提供模型
Model: HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335 Source: Original Platform
This commit is contained in:
37
.gitattributes
vendored
Normal file
37
.gitattributes
vendored
Normal file
@@ -0,0 +1,37 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
training_curves.png filter=lfs diff=lfs merge=lfs -text
|
||||
202
LICENSE
Normal file
202
LICENSE
Normal file
@@ -0,0 +1,202 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright 2024 Alibaba Cloud
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
314
README.md
Normal file
314
README.md
Normal file
@@ -0,0 +1,314 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
base_model: Qwen/Qwen3-30B-A3B
|
||||
tags:
|
||||
- qwen3
|
||||
- qwen3-moe
|
||||
- terminal-agent
|
||||
- reinforcement-learning
|
||||
- grpo
|
||||
- camel-ai
|
||||
- openclaw-rl
|
||||
- agent
|
||||
- terminal-bench
|
||||
- moe
|
||||
library_name: transformers
|
||||
language:
|
||||
- en
|
||||
- zh
|
||||
pipeline_tag: text-generation
|
||||
---
|
||||
|
||||
# Qwen3-30B-A3B-OpenClaw-RL-iter335
|
||||
|
||||
把 [`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B) (vanilla, 30B total / 3B active MoE) 用 [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) 的训练框架做 **GRPO outcome-only RL**(dense pass-rate reward,无 PRM、无 process reward)跑出来的 **best 中间 checkpoint**,对应 [wandb run `b0il0mq4`](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4) 的 rollout 335 (megatron iteration 335)。
|
||||
|
||||
在 [`terminal-bench` v0.1.x](https://github.com/laude-institute/terminal-bench/tree/v0.1.x) 66 个 OOD task 上把 base Qwen3-30B-A3B 的 pass@1 从 0.0606 推到 **0.0808** (+33%)、pass@3 从 0.1212 推到 **0.1515** (+25%),是这次 30B fresh run 训练中**唯一明确"超越自身 base"的 ckpt**。
|
||||
|
||||
> **完全 drop-in 兼容**:config / generation_config / tokenizer 全部跟上游 `Qwen/Qwen3-30B-A3B` 字节级一致;任何能跑 Qwen3-30B-A3B 的工具链(HF transformers、vLLM、sglang、ollama、llama.cpp 转 GGUF 等)**一行都不用改**。
|
||||
|
||||
完整的训练 + eval 详细分析见 [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14)。
|
||||
|
||||
---
|
||||
|
||||
## 目录
|
||||
|
||||
1. [快速开始](#快速开始)
|
||||
2. [模型从哪来](#模型从哪来)
|
||||
3. [训练设置](#训练设置)
|
||||
4. [训练曲线](#训练曲线)
|
||||
5. [评测结果](#评测结果)
|
||||
6. [为什么是 iter_335 而不是最新 ckpt](#为什么是-iter_335-而不是最新-ckpt)
|
||||
7. [推理时的 on-the-wire 协议](#推理时的-on-the-wire-协议)
|
||||
8. [已知限制](#已知限制)
|
||||
9. [复现 / 推理工具](#复现--推理工具)
|
||||
10. [引用 / 致谢](#引用--致谢)
|
||||
|
||||
---
|
||||
|
||||
## 快速开始
|
||||
|
||||
### 用 HF `transformers`
|
||||
|
||||
```python
|
||||
from transformers import AutoTokenizer, AutoModelForCausalLM
|
||||
|
||||
tok = AutoTokenizer.from_pretrained("HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335")
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
"HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335",
|
||||
torch_dtype="bfloat16",
|
||||
device_map="auto",
|
||||
)
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "Write a one-line bash command to recursively count *.py files under /usr/lib."}
|
||||
]
|
||||
text = tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
|
||||
inputs = tok(text, return_tensors="pt").to(model.device)
|
||||
out = model.generate(**inputs, max_new_tokens=256, temperature=0.2)
|
||||
print(tok.decode(out[0][inputs.input_ids.shape[1]:], skip_special_tokens=True))
|
||||
```
|
||||
|
||||
### 用 sglang(推荐,配 tool-call 支持 + MoE 加速)
|
||||
|
||||
```bash
|
||||
python -m sglang.launch_server \
|
||||
--model-path HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335 \
|
||||
--served-model-name qwen3-30b-rl-iter335 \
|
||||
--tp 4 --port 30000 \
|
||||
--mem-fraction-static 0.85 \
|
||||
--tool-call-parser qwen \
|
||||
--disable-custom-all-reduce
|
||||
```
|
||||
|
||||
跟 base Qwen3-30B-A3B 一样的启动参数。TP=4 时 30B MoE 单 rank ~15 GB params + ~120 GB kv cache,放 4× 80 GB GPU 或 4× 143 GB GPU 都没问题。`--disable-custom-all-reduce` 是 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 已知 Hopper bug 的兜底 flag,保留即可。`--tool-call-parser qwen`(或 `qwen25`)把模型的 `<tool_call>{...}</tool_call>` 还原成 OpenAI 结构化 `tool_calls`,这正是训练时 rollouts 用的 parser。
|
||||
|
||||
### 用 vLLM
|
||||
|
||||
```bash
|
||||
vllm serve HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335 \
|
||||
--tensor-parallel-size 4 \
|
||||
--enable-auto-tool-choice --tool-call-parser hermes
|
||||
```
|
||||
|
||||
### 用 oc-repl 直接交互(端到端 demo)
|
||||
|
||||
[`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) 是一个针对这个 family 量身写的 Codex 风格 REPL:
|
||||
|
||||
```bash
|
||||
pip install git+https://github.com/HansBug/oc-repl
|
||||
|
||||
# 起一个 sandbox 容器(任何 ubuntu/python image 都行)
|
||||
docker run -d --name oc-sandbox -w /app ubuntu:24.04 sleep infinity
|
||||
|
||||
# 启动交互式 REPL
|
||||
oc-repl --sandbox docker:oc-sandbox \
|
||||
--api-base http://127.0.0.1:30000/v1 \
|
||||
--model qwen3-30b-rl-iter335
|
||||
```
|
||||
|
||||
oc-repl 默认 `--protocol camel-terminal-toolkit` 对齐训练分布,详见仓库 README。
|
||||
|
||||
---
|
||||
|
||||
## 模型从哪来
|
||||
|
||||
`Qwen3-30B-A3B-OpenClaw-RL-iter335` = `Qwen3-30B-A3B` (vanilla, **不是** Coder-30B-A3B-Instruct,也**不是** 30B-A3B-Instruct-2507) + **335 个 RL rollout** 的 GRPO outcome-only 训练,rollout 在 [SETA terminal env pool](https://github.com/camel-ai/seta) 提供的 **1376 个 terminal task** 上。每个 task 是一个真实的 Linux shell 任务(修脚本、解 base64、起 server、跑 pytest…),在隔离 docker 容器里跑,agent 通过 [`camel.toolkits.TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) 暴露的 4 个工具操作 shell。
|
||||
|
||||
| | 值 |
|
||||
|---|---|
|
||||
| 基座 | [`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B)(**vanilla** MoE,30B total / 3B active,128 expert / topk=8 / 48 layer / hidden 2048)|
|
||||
| 训练框架 | [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) (Megatron + slime + sglang rollout) |
|
||||
| Agent harness | [camel-ai `ChatAgent` + `TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) |
|
||||
| 数据 | [seta_env](https://github.com/camel-ai/seta) 1376 task pool(含 [terminal-bench](https://github.com/laude-institute/terminal-bench) v0.1.x 86 task 子集) |
|
||||
| 算法 | GRPO outcome-only,dense pass-rate reward |
|
||||
| 训练 rollout 数 | 335 rollout(约 4× n_samples × prompt 数) |
|
||||
| 训练硬件 | 8× NVIDIA H200,actor 4×TP + 4×EP + ETP=1,rollout 4×TP |
|
||||
| 训练时长 | 约 4 天到 iter_335 完成(完整 fresh run 直到 iter_367 共 ~5.5 天 / 367 rollout)|
|
||||
| Wandb | [`hansbug/openclaw-terminal-rl/runs/b0il0mq4`](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4) |
|
||||
| Group | `qwen3-30b-a3b-vanilla-fresh-run1` |
|
||||
|
||||
**为什么是 vanilla 而不是 Coder-30B-A3B-Instruct?** 初次用 Coder-30B-A3B-Instruct 跑了 1114 个 rollout 全 fail —— 它的 `chat_template.jinja` 用 `tool_call.arguments|items` 要求 `arguments` 是 dict,但 slime/sglang 走 OpenAI 协议把 `arguments` 序列化成 JSON string → jinja2 `TypeError`。vanilla `Qwen3-30B-A3B` 的 chat_template 与 8B 字节级相同,整条 slime + terminus-2 + harbor 链路 0 修改通过。详见 [`HansBug/OpenClaw-RL` issue #14 §1.1](https://github.com/HansBug/OpenClaw-RL/issues/14)。
|
||||
|
||||
---
|
||||
|
||||
## 训练设置
|
||||
|
||||
完整启动命令见 [`run_qwen3_30b_a3b_experiment.sh`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/run_qwen3_30b_a3b_experiment.sh)。关键超参(与 8B run-3 算法侧完全相同,仅 MoE 并行差异):
|
||||
|
||||
```bash
|
||||
# GRPO 算法 (同 8B run-3, issue #4)
|
||||
--advantage-estimator grpo
|
||||
--use-kl-loss --kl-loss-coef 0.01 --kl-loss-type k3
|
||||
--dynamic_history
|
||||
|
||||
# Optimizer (low LR + Adam beta2=0.98)
|
||||
--optimizer adam --lr 1e-6 --lr-decay-style constant
|
||||
--weight-decay 0.1 --adam-beta1 0.9 --adam-beta2 0.98
|
||||
|
||||
# Rollout (同 8B)
|
||||
--rollout-batch-size 16
|
||||
--n-samples-per-prompt 8 # GRPO group size = 8
|
||||
--num-steps-per-rollout 2
|
||||
--rollout-temperature 1
|
||||
--rollout-max-response-len 8192
|
||||
--rollout-max-context-len 16384
|
||||
|
||||
# Megatron — MoE 并行(与 8B 不同)
|
||||
--tensor-model-parallel-size 4
|
||||
--expert-model-parallel-size 4 # 128 expert 分到 4 张 actor 卡,每卡 32 expert
|
||||
--expert-tensor-parallel-size 1
|
||||
--sequence-parallel
|
||||
--recompute-granularity full
|
||||
|
||||
# Save / freq
|
||||
--save-interval 16 # save 每 16 rollout(≈ 5.3h/save)
|
||||
--num-rollout 2000 # 训练目标(实际跑到 367 停)
|
||||
|
||||
# Rollout agent + tool-call parser
|
||||
--custom-config-path configs/rollout_qwen3.yaml # tool_call_parser: qwen25
|
||||
# max_iteration: 10
|
||||
```
|
||||
|
||||
Reward signal:**dense pass-rate, outcome-only** —— 每个 rollout 跑完后由任务自带的 `run-tests.sh` 在 sandbox 里跑 pytest,`reward = passed / total_tests ∈ [0, 1]`,再线性变换到 `score = 2·reward − 1 ∈ [-1, +1]` 喂给 GRPO;**没有 PRM / process reward**,整个 episode 共用一个标量。
|
||||
|
||||
`max_iteration: 10`:训练时 agent loop 上限 10 轮。
|
||||
|
||||
---
|
||||
|
||||
## 训练曲线
|
||||
|
||||
来自 [wandb run b0il0mq4](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4):
|
||||
|
||||

|
||||
|
||||
六个子图说明:
|
||||
|
||||
| 子图 | 解读 |
|
||||
|---|---|
|
||||
| **raw_reward** | rollout 阶段 reward 均值。-0.07 (rollout 0-49) → +0.21 (rollout 320-335) 单调上升。**穿越 0 出现在 rollout 50 左右**,即模型从"过的 test < 一半"转向"过的 test > 一半"。iter_335 这个 ckpt 对应的训练窗口 raw_reward 均值 **+0.21**,比 base 提升 ~0.28。|
|
||||
| **response_length** | 每轮 assistant 输出的平均 token 数。从 217 涨到 413,**单调上升、没有 plateau**——是 30B 远未到达容量上限的关键信号(vs 8B run-3 后期横盘 300-400)。|
|
||||
| **grad_norm** | 梯度幅度(log 尺度)。median 0.30-0.50 健康,期间 5 次单步 spike(最大 step 38 grad=1.36e8 / kl=5.6e5),全部被 megatron `--clip-grad 1.0` 兜底,下一步即恢复。|
|
||||
| **kl_loss** | K3 KL estimator。median 0.05-0.15 健康,远低于 0.5 阈值。spike 与 grad_norm 同步出现,clip-grad 同样吸收。|
|
||||
| **entropy_loss** | 0.13-0.17 持续,无 collapse。|
|
||||
| **pg_clipfrac** | PPO clip 比例 ~0.6%(合理)。|
|
||||
|
||||
详细的训练曲线、异常事件分析、vs 8B run-3 叠加图见 [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14)。
|
||||
|
||||
---
|
||||
|
||||
## 评测结果
|
||||
|
||||
### 在 terminal-bench v0.1.x(66-task harbor 子集)上的 pass@1 / pass@3
|
||||
|
||||
iter_335 这个 ckpt 在 OOD eval(agent harness 切到 `harbor run --agent terminus-2`,与 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 8B run-3 同协议)上的成绩:
|
||||
|
||||

|
||||
|
||||
| Model | pass@1 | pass@3 | unique solved (/66) | solved trials (/198) |
|
||||
|---|---:|---:|---:|---:|
|
||||
| `qwen3-30b-a3b-base` (vanilla) | 0.0606 | 0.1212 | 8 | 12 |
|
||||
| **`qwen3-30b-a3b-OpenClaw-RL-iter335`**(本 ckpt) | **0.0808** ✨ | **0.1515** ✨ | **10** | **16** |
|
||||
| `qwen3-30b-a3b-OpenClaw-RL-iter351` (later) | 0.0556 ⚠ | 0.0758 ⚠ | 5 | 11 |
|
||||
| `qwen3-30b-a3b-OpenClaw-RL-iter367` (final) | 0.0657 | 0.1212 | 8 | 13 |
|
||||
|
||||
**iter_335 是这次 fresh run 中唯一 pass@1 / pass@3 / unique_solved 三项全部超过 base 的 ckpt**。iter_351 在中间出现明显回落(甚至跌穿 base),iter_367 部分恢复但仍未回到 iter_335 峰。详细分析见 [issue #14 eval comment](https://github.com/HansBug/OpenClaw-RL/issues/14#issuecomment-4537976268)。
|
||||
|
||||
### vs 8B run-3([issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 同 66-task 标尺)
|
||||
|
||||

|
||||
|
||||
| Model | pass@1 | pass@3 | 体量 | 备注 |
|
||||
|---|---:|---:|---|---|
|
||||
| **Qwen3-30B-A3B-OpenClaw-RL-iter335**(本 ckpt) | **0.081** | **0.152** | 30B / 3B active MoE | **本 ckpt** |
|
||||
| Qwen3-30B-A3B (vanilla base) | 0.061 | 0.121 | 30B / 3B active | 训练起点 |
|
||||
| Qwen3-8B-OpenClaw-RL-iter215 ([issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8)) | 0.056 | 0.091 | 8B dense | 8B run-3 OOD 峰 |
|
||||
| Qwen3-8B (base) | 0.020 | 0.030 | 8B dense | 8B 起点 |
|
||||
| Qwen3-235B-A22B + Terminus 1 | 0.066 | n/a | 235B / 22B active | TB 官方 80-task |
|
||||
| DeepSeek-R1 + Terminus 1 | 0.057 | n/a | 671B / 37B active | TB 官方 80-task |
|
||||
| Qwen3-32B + TerminalAgent | 0.155 | n/a | 32B dense | TB 官方 80-task |
|
||||
| Claude 4.5 Sonnet | 0.645 | n/a | 闭源 | frontier |
|
||||
| GPT-5 | 0.525 | n/a | 闭源 | frontier |
|
||||
|
||||
> 注:本 ckpt 跑的是 66-task 子集([harbor migration 失败 20 task](https://github.com/HansBug/OpenClaw-RL/issues/8) 后剩下的可跑部分),与 leaderboard 80-task 全集**不可直接横比**;但唯一靠谱的同标尺对比是与 issue #8 同样 66 子集的 Qwen3-8B 数字,本 ckpt 的 pass@1 = 0.081 比 Qwen3-8B-iter215 (0.056) 高 **44%**。
|
||||
|
||||
### 任务级别的 OOD 解题分布
|
||||
|
||||

|
||||
|
||||
14 个 task 在所有 4 个 ckpt × 3 attempt 中至少被解出 1 次,其余 52 task 全部 0 solved(与 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 60/66 全 0 同 pattern)。iter_335 独占解出的 task:`csv-to-parquet`、`path-tracing-reverse`、`processing-pipeline 2x`。
|
||||
|
||||
---
|
||||
|
||||
## 为什么是 iter_335 而不是最新 ckpt
|
||||
|
||||
完整 30B fresh run 跑到 iter_367 停(rollout 367 ≈ 训练 4.5 天),按 retention daemon 配置仅保留 latest 3 ckpt:iter_335 / iter_351 / iter_367。所有 4 个候选(含 base)都在 TB v0.1.x 跑过 OOD eval,得到的训练→OOD 轨迹是:
|
||||
|
||||
| ckpt | rollout | in-domain raw_reward (训练侧) | OOD pass@1 | 训练→OOD 是否对齐? |
|
||||
|---|---:|---:|---:|---|
|
||||
| base | — | — | 0.0606 | — |
|
||||
| **iter_335** | **335** | **+0.21** | **0.0808** ✨ | ✅ 训练 ↑ + OOD ↑ |
|
||||
| iter_351 | 351 | +0.18(含全局峰 r346 +0.5651) | 0.0556 | ❌ 训练保持高位但 OOD 跌穿 base |
|
||||
| iter_367 | 367 | +0.232(最高窗口均值) | 0.0657 | ❌ 训练 ↑ 但 OOD 没跟上 |
|
||||
|
||||
iter_335 是这次训练**唯一明确"训练侧 ↑ + OOD ↑"的 ckpt**。这是 [issue #8 iter215](https://github.com/HansBug/OpenClaw-RL/issues/8)(8B run-3 也在中段 ckpt 拿 OOD 峰,最后回落)同款现象。如果你要拿一个最有可能在新 OOD task 上 work 的 30B fresh-run ckpt,**应该用这个 iter_335**,不要用最新的 iter_367。
|
||||
|
||||
---
|
||||
|
||||
## 推理时的 on-the-wire 协议
|
||||
|
||||
**训练时这个 ckpt 看到的协议是 [camel-ai `TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) 的 4 工具 OpenAI function-calling**(与 [`Qwen3-8B-OpenClaw-RL-iter215`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) 完全相同),**不是** terminal-bench 的 `terminus-2` JSON 协议。
|
||||
|
||||
| 文件 | 说明 |
|
||||
|---|---|
|
||||
| [`terminal-rl/agent/camel_agent.py`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/agent/camel_agent.py) | rollout agent = `camel.agents.ChatAgent` 子类 |
|
||||
| [`terminal-rl/remote/terminal_env.py`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/remote/terminal_env.py) | env 在 reset 时把 `camel.toolkits.TerminalToolkit` 的 4 个工具(`shell_exec` / `shell_view` / `shell_write_to_process` / `shell_write_content_to_file`)封装成 OpenAI tool schema 喂给 rollout |
|
||||
| [`terminal-rl/configs/rollout_qwen3.yaml`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/configs/rollout_qwen3.yaml) | `tool_call_parser: qwen25` — sglang 把模型 `<tool_call>{...}</tool_call>` markup 转回 OpenAI tool_calls |
|
||||
|
||||
System prompt 是 [`terminal-rl/agent/camel_agent.py::get_developer_agent_prompt()`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/agent/camel_agent.py) 在 `system='Linux (in Docker)', machine='x86_64', non_think_mode=True` 配置下产出,结尾加 `/no_think` 关闭 qwen3 的 think trace。
|
||||
|
||||
如果你想 100% 复刻训练分布在 inference 用这个 ckpt,用 [`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) 的默认 `--protocol camel-terminal-toolkit`。
|
||||
|
||||
---
|
||||
|
||||
## 已知限制
|
||||
|
||||
1. **30B 在 OOD 上的 RL 边际收益比 8B 小**。本 ckpt vs 30B base:pass@1 +33%;[Qwen3-8B-iter215](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) vs Qwen3-8B base 是 +180%。如果你的目标是 frontier OOD 性能,纯 RL 已经 saturating,需要换 cold-start SFT。这与 [issue #11 容量假说](https://github.com/HansBug/OpenClaw-RL/issues/11) 一致。
|
||||
|
||||
2. **iter_351 / iter_367 的存在说明 RL 训练后期有 OOD 漂移风险**。iter_335 是这次训练全 4 ckpt 中唯一 net-positive 的;更晚的 iter_351 反而跌穿 base。如果你打算复现这个训练 pipeline,建议把 ckpt freeze 频率提高,每 50-100 rollout 都做 OOD eval 看 trajectory。
|
||||
|
||||
3. **模型不会主动停止 tool_call**。训练时 agent loop 由 `max_iteration: 10` 硬截断 + outcome reward 兜底,inference 时要自己设上限([oc-repl 默认 12 轮](https://github.com/HansBug/oc-repl/blob/main/src/oc_repl/engine.py))。
|
||||
|
||||
4. **复杂多步 task adherence 中等**。简单任务(chmod、cat、ls)adherence 满分;多文件 Python 服务器、复杂 awk 这种偶尔会 thinking 太久没产出 tool_call。
|
||||
|
||||
5. **terminus-2 / terminus-XML 协议是 OOD**。模型没在这些协议上 RL 训练过,能不能跑通靠 qwen3 底座的通用指令跟随能力。要复刻训练分布请用 camel TerminalToolkit 协议。
|
||||
|
||||
6. **不是 frontier 水平**。pass@1 0.081 跟 Qwen3-235B 同水位(甚至略高),但跟 Claude 4.5 / GPT-5 还有 6-8× 差距。这个 ckpt 适合做 RL 训练框架的 baseline / case study / MoE base 上 RL 的对比基准,不适合直接当 production agent 用。
|
||||
|
||||
---
|
||||
|
||||
## 复现 / 推理工具
|
||||
|
||||
| 仓库 / 工具 | 用途 |
|
||||
|---|---|
|
||||
| [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) | 完整的训练框架(slime + Megatron + sglang),含 launch script、配置、agent code |
|
||||
| [`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) | 针对这个 family ckpt 的 Codex 风格 REPL,支持 4 种推理协议 (`camel-terminal-toolkit` 默认 = 训练分布字节级复刻) |
|
||||
| [`HansBug/Qwen3-8B-OpenClaw-RL-iter215`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) | **同 family** 的 8B ckpt(base = Qwen3-8B),用于 same-protocol 同标尺横比 |
|
||||
| [`HansBug/Qwen3-8B-OpenClaw-RL-tboverfit-iter311`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-tboverfit-iter311) | 同 family 的 8B eval-as-train 上界 probe ckpt |
|
||||
| [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14) | 本 ckpt 的完整训练 + eval 报告(118h 训练曲线 + 4 ckpt OOD eval + vs 8B 对比) |
|
||||
| [`HansBug/OpenClaw-RL` issue #4](https://github.com/HansBug/OpenClaw-RL/issues/4) | 8B run-3 训练报告(同算法 / 同 dataset / 同 lr,仅 base 模型不同) |
|
||||
| [`HansBug/OpenClaw-RL` issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) | 8B run-3 在 TB v0.1.x 66-task 的同标尺 OOD eval(本 ckpt eval 完全复用了该协议) |
|
||||
|
||||
---
|
||||
|
||||
## 引用 / 致谢
|
||||
|
||||
- 基座:[`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B)(Apache-2.0,vanilla 版,2025-04 release)
|
||||
- 训练框架:[`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL)
|
||||
- Agent harness:[`camel-ai/camel`](https://github.com/camel-ai/camel) [`TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit)
|
||||
- Eval benchmark:[`laude-institute/terminal-bench`](https://github.com/laude-institute/terminal-bench)
|
||||
- Rollout 训练数据 pool:[`camel-ai/seta`](https://github.com/camel-ai/seta) 1376 task
|
||||
|
||||
License: Apache-2.0(继承自 Qwen3-30B-A3B)。
|
||||
|
||||
如果你用这个 ckpt 做 paper / blog post,欢迎引用 [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) 仓库 + 这个 model card。
|
||||
38
config.json
Normal file
38
config.json
Normal file
@@ -0,0 +1,38 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3MoeForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 151643,
|
||||
"decoder_sparse_step": 1,
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 6144,
|
||||
"max_position_embeddings": 40960,
|
||||
"max_window_layers": 48,
|
||||
"mlp_only_layers": [],
|
||||
"model_type": "qwen3_moe",
|
||||
"moe_intermediate_size": 768,
|
||||
"norm_topk_prob": true,
|
||||
"num_attention_heads": 32,
|
||||
"num_experts": 128,
|
||||
"num_experts_per_tok": 8,
|
||||
"num_hidden_layers": 48,
|
||||
"num_key_value_heads": 4,
|
||||
"output_router_logits": false,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 1000000.0,
|
||||
"router_aux_loss_coef": 0.001,
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": false,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.51.0",
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
BIN
eval_pass_rate.png
Normal file
BIN
eval_pass_rate.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 52 KiB |
13
generation_config.json
Normal file
13
generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"bos_token_id": 151643,
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151645,
|
||||
151643
|
||||
],
|
||||
"pad_token_id": 151643,
|
||||
"temperature": 0.6,
|
||||
"top_k": 20,
|
||||
"top_p": 0.95,
|
||||
"transformers_version": "4.51.0"
|
||||
}
|
||||
BIN
leaderboard.png
Normal file
BIN
leaderboard.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 46 KiB |
151388
merges.txt
Normal file
151388
merges.txt
Normal file
File diff suppressed because it is too large
Load Diff
3
model-00000-of-00012.safetensors
Normal file
3
model-00000-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d87654f409ae0c2daf1bbf8d52aad7796f973de28e39385dab5bece8f80df646
|
||||
size 5368403824
|
||||
3
model-00001-of-00012.safetensors
Normal file
3
model-00001-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2ace3b84fae9d386cde1bfeafbff6643ab4821db2697f6c04ac2126e8b1988a6
|
||||
size 5366338784
|
||||
3
model-00002-of-00012.safetensors
Normal file
3
model-00002-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:dcad34d518774682dcd0cc69ef11961547872e2abe6457d9fcae85a25f288b8e
|
||||
size 5365807096
|
||||
3
model-00003-of-00012.safetensors
Normal file
3
model-00003-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0e1e77f2cb4f880cb55e383d24bf8946e8704c25be3ccbccb9f89007c083cfca
|
||||
size 5365807912
|
||||
3
model-00004-of-00012.safetensors
Normal file
3
model-00004-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:773de90228818b49c2e90107c7f571a8a6bb4352640cfcec34de7f2340a9ec97
|
||||
size 5366340528
|
||||
3
model-00005-of-00012.safetensors
Normal file
3
model-00005-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d3dd1923ee40bceda9c58cd1f4c29af7fe9bd5c64e059915e2189dbd8e532ce0
|
||||
size 5365807832
|
||||
3
model-00006-of-00012.safetensors
Normal file
3
model-00006-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0e44b393301da6a6e77ea28bd5edacbb8a5255999f1fabb828b5b71c823274f0
|
||||
size 5365807888
|
||||
3
model-00007-of-00012.safetensors
Normal file
3
model-00007-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:034830e720f81bd8756538629da33b20b377f64a4463205c06028d040819b771
|
||||
size 5365807968
|
||||
3
model-00008-of-00012.safetensors
Normal file
3
model-00008-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ddf4a72a4efd9bea144694b509598a312280e5ada55dedd5b2d54611bc38c50d
|
||||
size 5366340408
|
||||
3
model-00009-of-00012.safetensors
Normal file
3
model-00009-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:26e9bb6013595576fe90d0316b3a33d4ccfb8d02acb4c39d3c3d9b0cb65aba89
|
||||
size 5365807856
|
||||
3
model-00010-of-00012.safetensors
Normal file
3
model-00010-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ad77a88f4bc90b14112bc51315360faa6c1dd75ffa9fe75339abeefaadfa95b1
|
||||
size 5365807960
|
||||
3
model-00011-of-00012.safetensors
Normal file
3
model-00011-of-00012.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:e0727dee24c5c05100124a647c1fb746a041c1a95db56c916e5b32394a2c921c
|
||||
size 2038500104
|
||||
18874
model.safetensors.index.json
Normal file
18874
model.safetensors.index.json
Normal file
File diff suppressed because it is too large
Load Diff
BIN
per_task_heatmap.png
Normal file
BIN
per_task_heatmap.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 81 KiB |
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
BIN
tokenizer.json
(Stored with Git LFS)
Normal file
Binary file not shown.
239
tokenizer_config.json
Normal file
239
tokenizer_config.json
Normal file
@@ -0,0 +1,239 @@
|
||||
{
|
||||
"add_bos_token": false,
|
||||
"add_prefix_space": false,
|
||||
"added_tokens_decoder": {
|
||||
"151643": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151644": {
|
||||
"content": "<|im_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151645": {
|
||||
"content": "<|im_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151646": {
|
||||
"content": "<|object_ref_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151647": {
|
||||
"content": "<|object_ref_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151648": {
|
||||
"content": "<|box_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151649": {
|
||||
"content": "<|box_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151650": {
|
||||
"content": "<|quad_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151651": {
|
||||
"content": "<|quad_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151652": {
|
||||
"content": "<|vision_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151653": {
|
||||
"content": "<|vision_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151654": {
|
||||
"content": "<|vision_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151655": {
|
||||
"content": "<|image_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151656": {
|
||||
"content": "<|video_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151657": {
|
||||
"content": "<tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151658": {
|
||||
"content": "</tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151659": {
|
||||
"content": "<|fim_prefix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151660": {
|
||||
"content": "<|fim_middle|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151661": {
|
||||
"content": "<|fim_suffix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151662": {
|
||||
"content": "<|fim_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151663": {
|
||||
"content": "<|repo_name|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151664": {
|
||||
"content": "<|file_sep|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151665": {
|
||||
"content": "<tool_response>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151666": {
|
||||
"content": "</tool_response>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151667": {
|
||||
"content": "<think>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151668": {
|
||||
"content": "</think>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
}
|
||||
},
|
||||
"additional_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"bos_token": null,
|
||||
"chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {{- messages[0].content + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0].content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if message.content is string %}\n {%- set content = message.content %}\n {%- else %}\n {%- set content = '' %}\n {%- endif %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- if loop.index0 > ns.last_query_index %}\n {%- if loop.last or (not loop.last and reasoning_content) %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content.strip('\\n') + '\\n</think>\\n\\n' + content.lstrip('\\n') }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '<tool_call>\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- endif %}\n{%- endif %}",
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|im_end|>",
|
||||
"errors": "replace",
|
||||
"model_max_length": 131072,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
3
training_curves.png
Normal file
3
training_curves.png
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ae8651c73c689c25cbf5ea9014c0e9673ecaac6d5463ad32ec173574d0485398
|
||||
size 391498
|
||||
1
vocab.json
Normal file
1
vocab.json
Normal file
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user