初始化项目,由ModelHub XC社区提供模型

Model: HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-26 19:24:01 +08:00
commit f1e869b047
26 changed files with 171148 additions and 0 deletions

37
.gitattributes vendored Normal file
View File

@@ -0,0 +1,37 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tokenizer.json filter=lfs diff=lfs merge=lfs -text
training_curves.png filter=lfs diff=lfs merge=lfs -text

202
LICENSE Normal file
View File

@@ -0,0 +1,202 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright 2024 Alibaba Cloud
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.

314
README.md Normal file
View File

@@ -0,0 +1,314 @@
---
license: apache-2.0
base_model: Qwen/Qwen3-30B-A3B
tags:
- qwen3
- qwen3-moe
- terminal-agent
- reinforcement-learning
- grpo
- camel-ai
- openclaw-rl
- agent
- terminal-bench
- moe
library_name: transformers
language:
- en
- zh
pipeline_tag: text-generation
---
# Qwen3-30B-A3B-OpenClaw-RL-iter335
把 [`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B) (vanilla, 30B total / 3B active MoE) 用 [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) 的训练框架做 **GRPO outcome-only RL**(dense pass-rate reward,无 PRM、无 process reward)跑出来的 **best 中间 checkpoint**,对应 [wandb run `b0il0mq4`](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4) 的 rollout 335 (megatron iteration 335)。
在 [`terminal-bench` v0.1.x](https://github.com/laude-institute/terminal-bench/tree/v0.1.x) 66 个 OOD task 上把 base Qwen3-30B-A3B 的 pass@1 从 0.0606 推到 **0.0808** (+33%)、pass@3 从 0.1212 推到 **0.1515** (+25%),是这次 30B fresh run 训练中**唯一明确"超越自身 base"的 ckpt**。
> **完全 drop-in 兼容**:config / generation_config / tokenizer 全部跟上游 `Qwen/Qwen3-30B-A3B` 字节级一致;任何能跑 Qwen3-30B-A3B 的工具链(HF transformers、vLLM、sglang、ollama、llama.cpp 转 GGUF 等)**一行都不用改**。
完整的训练 + eval 详细分析见 [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14)。
---
## 目录
1. [快速开始](#快速开始)
2. [模型从哪来](#模型从哪来)
3. [训练设置](#训练设置)
4. [训练曲线](#训练曲线)
5. [评测结果](#评测结果)
6. [为什么是 iter_335 而不是最新 ckpt](#为什么是-iter_335-而不是最新-ckpt)
7. [推理时的 on-the-wire 协议](#推理时的-on-the-wire-协议)
8. [已知限制](#已知限制)
9. [复现 / 推理工具](#复现--推理工具)
10. [引用 / 致谢](#引用--致谢)
---
## 快速开始
### 用 HF `transformers`
```python
from transformers import AutoTokenizer, AutoModelForCausalLM
tok = AutoTokenizer.from_pretrained("HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335")
model = AutoModelForCausalLM.from_pretrained(
"HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335",
torch_dtype="bfloat16",
device_map="auto",
)
messages = [
{"role": "user", "content": "Write a one-line bash command to recursively count *.py files under /usr/lib."}
]
text = tok.apply_chat_template(messages, tokenize=False, add_generation_prompt=True)
inputs = tok(text, return_tensors="pt").to(model.device)
out = model.generate(**inputs, max_new_tokens=256, temperature=0.2)
print(tok.decode(out[0][inputs.input_ids.shape[1]:], skip_special_tokens=True))
```
### 用 sglang(推荐,配 tool-call 支持 + MoE 加速)
```bash
python -m sglang.launch_server \
--model-path HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335 \
--served-model-name qwen3-30b-rl-iter335 \
--tp 4 --port 30000 \
--mem-fraction-static 0.85 \
--tool-call-parser qwen \
--disable-custom-all-reduce
```
跟 base Qwen3-30B-A3B 一样的启动参数。TP=4 时 30B MoE 单 rank ~15 GB params + ~120 GB kv cache,放 4× 80 GB GPU 或 4× 143 GB GPU 都没问题。`--disable-custom-all-reduce` 是 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 已知 Hopper bug 的兜底 flag,保留即可。`--tool-call-parser qwen`(或 `qwen25`)把模型的 `<tool_call>{...}</tool_call>` 还原成 OpenAI 结构化 `tool_calls`,这正是训练时 rollouts 用的 parser。
### 用 vLLM
```bash
vllm serve HansBug/Qwen3-30B-A3B-OpenClaw-RL-iter335 \
--tensor-parallel-size 4 \
--enable-auto-tool-choice --tool-call-parser hermes
```
### 用 oc-repl 直接交互(端到端 demo)
[`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) 是一个针对这个 family 量身写的 Codex 风格 REPL:
```bash
pip install git+https://github.com/HansBug/oc-repl
# 起一个 sandbox 容器(任何 ubuntu/python image 都行)
docker run -d --name oc-sandbox -w /app ubuntu:24.04 sleep infinity
# 启动交互式 REPL
oc-repl --sandbox docker:oc-sandbox \
--api-base http://127.0.0.1:30000/v1 \
--model qwen3-30b-rl-iter335
```
oc-repl 默认 `--protocol camel-terminal-toolkit` 对齐训练分布,详见仓库 README。
---
## 模型从哪来
`Qwen3-30B-A3B-OpenClaw-RL-iter335` = `Qwen3-30B-A3B` (vanilla, **不是** Coder-30B-A3B-Instruct,也**不是** 30B-A3B-Instruct-2507) + **335 个 RL rollout** 的 GRPO outcome-only 训练,rollout 在 [SETA terminal env pool](https://github.com/camel-ai/seta) 提供的 **1376 个 terminal task** 上。每个 task 是一个真实的 Linux shell 任务(修脚本、解 base64、起 server、跑 pytest…),在隔离 docker 容器里跑,agent 通过 [`camel.toolkits.TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) 暴露的 4 个工具操作 shell。
| | 值 |
|---|---|
| 基座 | [`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B)(**vanilla** MoE,30B total / 3B active,128 expert / topk=8 / 48 layer / hidden 2048)|
| 训练框架 | [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) (Megatron + slime + sglang rollout) |
| Agent harness | [camel-ai `ChatAgent` + `TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) |
| 数据 | [seta_env](https://github.com/camel-ai/seta) 1376 task pool(含 [terminal-bench](https://github.com/laude-institute/terminal-bench) v0.1.x 86 task 子集) |
| 算法 | GRPO outcome-only,dense pass-rate reward |
| 训练 rollout 数 | 335 rollout(约 4× n_samples × prompt 数) |
| 训练硬件 | 8× NVIDIA H200,actor 4×TP + 4×EP + ETP=1,rollout 4×TP |
| 训练时长 | 约 4 天到 iter_335 完成(完整 fresh run 直到 iter_367 共 ~5.5 天 / 367 rollout)|
| Wandb | [`hansbug/openclaw-terminal-rl/runs/b0il0mq4`](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4) |
| Group | `qwen3-30b-a3b-vanilla-fresh-run1` |
**为什么是 vanilla 而不是 Coder-30B-A3B-Instruct?** 初次用 Coder-30B-A3B-Instruct 跑了 1114 个 rollout 全 fail —— 它的 `chat_template.jinja` 用 `tool_call.arguments|items` 要求 `arguments` 是 dict,但 slime/sglang 走 OpenAI 协议把 `arguments` 序列化成 JSON string → jinja2 `TypeError`。vanilla `Qwen3-30B-A3B` 的 chat_template 与 8B 字节级相同,整条 slime + terminus-2 + harbor 链路 0 修改通过。详见 [`HansBug/OpenClaw-RL` issue #14 §1.1](https://github.com/HansBug/OpenClaw-RL/issues/14)。
---
## 训练设置
完整启动命令见 [`run_qwen3_30b_a3b_experiment.sh`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/run_qwen3_30b_a3b_experiment.sh)。关键超参(与 8B run-3 算法侧完全相同,仅 MoE 并行差异):
```bash
# GRPO 算法 (同 8B run-3, issue #4)
--advantage-estimator grpo
--use-kl-loss --kl-loss-coef 0.01 --kl-loss-type k3
--dynamic_history
# Optimizer (low LR + Adam beta2=0.98)
--optimizer adam --lr 1e-6 --lr-decay-style constant
--weight-decay 0.1 --adam-beta1 0.9 --adam-beta2 0.98
# Rollout (同 8B)
--rollout-batch-size 16
--n-samples-per-prompt 8 # GRPO group size = 8
--num-steps-per-rollout 2
--rollout-temperature 1
--rollout-max-response-len 8192
--rollout-max-context-len 16384
# Megatron — MoE 并行(与 8B 不同)
--tensor-model-parallel-size 4
--expert-model-parallel-size 4 # 128 expert 分到 4 张 actor 卡,每卡 32 expert
--expert-tensor-parallel-size 1
--sequence-parallel
--recompute-granularity full
# Save / freq
--save-interval 16 # save 每 16 rollout(≈ 5.3h/save)
--num-rollout 2000 # 训练目标(实际跑到 367 停)
# Rollout agent + tool-call parser
--custom-config-path configs/rollout_qwen3.yaml # tool_call_parser: qwen25
# max_iteration: 10
```
Reward signal:**dense pass-rate, outcome-only** —— 每个 rollout 跑完后由任务自带的 `run-tests.sh` 在 sandbox 里跑 pytest,`reward = passed / total_tests ∈ [0, 1]`,再线性变换到 `score = 2·reward − 1 ∈ [-1, +1]` 喂给 GRPO;**没有 PRM / process reward**,整个 episode 共用一个标量。
`max_iteration: 10`:训练时 agent loop 上限 10 轮。
---
## 训练曲线
来自 [wandb run b0il0mq4](https://wandb.ai/hansbug/openclaw-terminal-rl/runs/b0il0mq4):
![training curves](training_curves.png)
六个子图说明:
| 子图 | 解读 |
|---|---|
| **raw_reward** | rollout 阶段 reward 均值。-0.07 (rollout 0-49) → +0.21 (rollout 320-335) 单调上升。**穿越 0 出现在 rollout 50 左右**,即模型从"过的 test < 一半"转向"过的 test > 一半"。iter_335 这个 ckpt 对应的训练窗口 raw_reward 均值 **+0.21**,比 base 提升 ~0.28。|
| **response_length** | 每轮 assistant 输出的平均 token 数。从 217 涨到 413,**单调上升、没有 plateau**——是 30B 远未到达容量上限的关键信号(vs 8B run-3 后期横盘 300-400)。|
| **grad_norm** | 梯度幅度(log 尺度)。median 0.30-0.50 健康,期间 5 次单步 spike(最大 step 38 grad=1.36e8 / kl=5.6e5),全部被 megatron `--clip-grad 1.0` 兜底,下一步即恢复。|
| **kl_loss** | K3 KL estimator。median 0.05-0.15 健康,远低于 0.5 阈值。spike 与 grad_norm 同步出现,clip-grad 同样吸收。|
| **entropy_loss** | 0.13-0.17 持续,无 collapse。|
| **pg_clipfrac** | PPO clip 比例 ~0.6%(合理)。|
详细的训练曲线、异常事件分析、vs 8B run-3 叠加图见 [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14)。
---
## 评测结果
### 在 terminal-bench v0.1.x(66-task harbor 子集)上的 pass@1 / pass@3
iter_335 这个 ckpt 在 OOD eval(agent harness 切到 `harbor run --agent terminus-2`,与 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 8B run-3 同协议)上的成绩:
![eval pass rate](eval_pass_rate.png)
| Model | pass@1 | pass@3 | unique solved (/66) | solved trials (/198) |
|---|---:|---:|---:|---:|
| `qwen3-30b-a3b-base` (vanilla) | 0.0606 | 0.1212 | 8 | 12 |
| **`qwen3-30b-a3b-OpenClaw-RL-iter335`**(本 ckpt) | **0.0808** ✨ | **0.1515** ✨ | **10** | **16** |
| `qwen3-30b-a3b-OpenClaw-RL-iter351` (later) | 0.0556 ⚠ | 0.0758 ⚠ | 5 | 11 |
| `qwen3-30b-a3b-OpenClaw-RL-iter367` (final) | 0.0657 | 0.1212 | 8 | 13 |
**iter_335 是这次 fresh run 中唯一 pass@1 / pass@3 / unique_solved 三项全部超过 base 的 ckpt**。iter_351 在中间出现明显回落(甚至跌穿 base),iter_367 部分恢复但仍未回到 iter_335 峰。详细分析见 [issue #14 eval comment](https://github.com/HansBug/OpenClaw-RL/issues/14#issuecomment-4537976268)。
### vs 8B run-3([issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 同 66-task 标尺)
![leaderboard](leaderboard.png)
| Model | pass@1 | pass@3 | 体量 | 备注 |
|---|---:|---:|---|---|
| **Qwen3-30B-A3B-OpenClaw-RL-iter335**(本 ckpt) | **0.081** | **0.152** | 30B / 3B active MoE | **本 ckpt** |
| Qwen3-30B-A3B (vanilla base) | 0.061 | 0.121 | 30B / 3B active | 训练起点 |
| Qwen3-8B-OpenClaw-RL-iter215 ([issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8)) | 0.056 | 0.091 | 8B dense | 8B run-3 OOD 峰 |
| Qwen3-8B (base) | 0.020 | 0.030 | 8B dense | 8B 起点 |
| Qwen3-235B-A22B + Terminus 1 | 0.066 | n/a | 235B / 22B active | TB 官方 80-task |
| DeepSeek-R1 + Terminus 1 | 0.057 | n/a | 671B / 37B active | TB 官方 80-task |
| Qwen3-32B + TerminalAgent | 0.155 | n/a | 32B dense | TB 官方 80-task |
| Claude 4.5 Sonnet | 0.645 | n/a | 闭源 | frontier |
| GPT-5 | 0.525 | n/a | 闭源 | frontier |
> 注:本 ckpt 跑的是 66-task 子集([harbor migration 失败 20 task](https://github.com/HansBug/OpenClaw-RL/issues/8) 后剩下的可跑部分),与 leaderboard 80-task 全集**不可直接横比**;但唯一靠谱的同标尺对比是与 issue #8 同样 66 子集的 Qwen3-8B 数字,本 ckpt 的 pass@1 = 0.081 比 Qwen3-8B-iter215 (0.056) 高 **44%**。
### 任务级别的 OOD 解题分布
![per-task heatmap](per_task_heatmap.png)
14 个 task 在所有 4 个 ckpt × 3 attempt 中至少被解出 1 次,其余 52 task 全部 0 solved(与 [issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) 60/66 全 0 同 pattern)。iter_335 独占解出的 task:`csv-to-parquet`、`path-tracing-reverse`、`processing-pipeline 2x`。
---
## 为什么是 iter_335 而不是最新 ckpt
完整 30B fresh run 跑到 iter_367 停(rollout 367 ≈ 训练 4.5 天),按 retention daemon 配置仅保留 latest 3 ckpt:iter_335 / iter_351 / iter_367。所有 4 个候选(含 base)都在 TB v0.1.x 跑过 OOD eval,得到的训练→OOD 轨迹是:
| ckpt | rollout | in-domain raw_reward (训练侧) | OOD pass@1 | 训练→OOD 是否对齐? |
|---|---:|---:|---:|---|
| base | — | — | 0.0606 | — |
| **iter_335** | **335** | **+0.21** | **0.0808** ✨ | ✅ 训练 ↑ + OOD ↑ |
| iter_351 | 351 | +0.18(含全局峰 r346 +0.5651) | 0.0556 | ❌ 训练保持高位但 OOD 跌穿 base |
| iter_367 | 367 | +0.232(最高窗口均值) | 0.0657 | ❌ 训练 ↑ 但 OOD 没跟上 |
iter_335 是这次训练**唯一明确"训练侧 ↑ + OOD ↑"的 ckpt**。这是 [issue #8 iter215](https://github.com/HansBug/OpenClaw-RL/issues/8)(8B run-3 也在中段 ckpt 拿 OOD 峰,最后回落)同款现象。如果你要拿一个最有可能在新 OOD task 上 work 的 30B fresh-run ckpt,**应该用这个 iter_335**,不要用最新的 iter_367。
---
## 推理时的 on-the-wire 协议
**训练时这个 ckpt 看到的协议是 [camel-ai `TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit) 的 4 工具 OpenAI function-calling**(与 [`Qwen3-8B-OpenClaw-RL-iter215`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) 完全相同),**不是** terminal-bench 的 `terminus-2` JSON 协议。
| 文件 | 说明 |
|---|---|
| [`terminal-rl/agent/camel_agent.py`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/agent/camel_agent.py) | rollout agent = `camel.agents.ChatAgent` 子类 |
| [`terminal-rl/remote/terminal_env.py`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/remote/terminal_env.py) | env 在 reset 时把 `camel.toolkits.TerminalToolkit` 的 4 个工具(`shell_exec` / `shell_view` / `shell_write_to_process` / `shell_write_content_to_file`)封装成 OpenAI tool schema 喂给 rollout |
| [`terminal-rl/configs/rollout_qwen3.yaml`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/configs/rollout_qwen3.yaml) | `tool_call_parser: qwen25` — sglang 把模型 `<tool_call>{...}</tool_call>` markup 转回 OpenAI tool_calls |
System prompt 是 [`terminal-rl/agent/camel_agent.py::get_developer_agent_prompt()`](https://github.com/HansBug/OpenClaw-RL/blob/main/terminal-rl/agent/camel_agent.py) 在 `system='Linux (in Docker)', machine='x86_64', non_think_mode=True` 配置下产出,结尾加 `/no_think` 关闭 qwen3 的 think trace。
如果你想 100% 复刻训练分布在 inference 用这个 ckpt,用 [`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) 的默认 `--protocol camel-terminal-toolkit`。
---
## 已知限制
1. **30B 在 OOD 上的 RL 边际收益比 8B 小**。本 ckpt vs 30B base:pass@1 +33%;[Qwen3-8B-iter215](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) vs Qwen3-8B base 是 +180%。如果你的目标是 frontier OOD 性能,纯 RL 已经 saturating,需要换 cold-start SFT。这与 [issue #11 容量假说](https://github.com/HansBug/OpenClaw-RL/issues/11) 一致。
2. **iter_351 / iter_367 的存在说明 RL 训练后期有 OOD 漂移风险**。iter_335 是这次训练全 4 ckpt 中唯一 net-positive 的;更晚的 iter_351 反而跌穿 base。如果你打算复现这个训练 pipeline,建议把 ckpt freeze 频率提高,每 50-100 rollout 都做 OOD eval 看 trajectory。
3. **模型不会主动停止 tool_call**。训练时 agent loop 由 `max_iteration: 10` 硬截断 + outcome reward 兜底,inference 时要自己设上限([oc-repl 默认 12 轮](https://github.com/HansBug/oc-repl/blob/main/src/oc_repl/engine.py))。
4. **复杂多步 task adherence 中等**。简单任务(chmod、cat、ls)adherence 满分;多文件 Python 服务器、复杂 awk 这种偶尔会 thinking 太久没产出 tool_call。
5. **terminus-2 / terminus-XML 协议是 OOD**。模型没在这些协议上 RL 训练过,能不能跑通靠 qwen3 底座的通用指令跟随能力。要复刻训练分布请用 camel TerminalToolkit 协议。
6. **不是 frontier 水平**。pass@1 0.081 跟 Qwen3-235B 同水位(甚至略高),但跟 Claude 4.5 / GPT-5 还有 6-8× 差距。这个 ckpt 适合做 RL 训练框架的 baseline / case study / MoE base 上 RL 的对比基准,不适合直接当 production agent 用。
---
## 复现 / 推理工具
| 仓库 / 工具 | 用途 |
|---|---|
| [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) | 完整的训练框架(slime + Megatron + sglang),含 launch script、配置、agent code |
| [`HansBug/oc-repl`](https://github.com/HansBug/oc-repl) | 针对这个 family ckpt 的 Codex 风格 REPL,支持 4 种推理协议 (`camel-terminal-toolkit` 默认 = 训练分布字节级复刻) |
| [`HansBug/Qwen3-8B-OpenClaw-RL-iter215`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-iter215) | **同 family** 的 8B ckpt(base = Qwen3-8B),用于 same-protocol 同标尺横比 |
| [`HansBug/Qwen3-8B-OpenClaw-RL-tboverfit-iter311`](https://huggingface.co/HansBug/Qwen3-8B-OpenClaw-RL-tboverfit-iter311) | 同 family 的 8B eval-as-train 上界 probe ckpt |
| [`HansBug/OpenClaw-RL` issue #14](https://github.com/HansBug/OpenClaw-RL/issues/14) | 本 ckpt 的完整训练 + eval 报告(118h 训练曲线 + 4 ckpt OOD eval + vs 8B 对比) |
| [`HansBug/OpenClaw-RL` issue #4](https://github.com/HansBug/OpenClaw-RL/issues/4) | 8B run-3 训练报告(同算法 / 同 dataset / 同 lr,仅 base 模型不同) |
| [`HansBug/OpenClaw-RL` issue #8](https://github.com/HansBug/OpenClaw-RL/issues/8) | 8B run-3 在 TB v0.1.x 66-task 的同标尺 OOD eval(本 ckpt eval 完全复用了该协议) |
---
## 引用 / 致谢
- 基座:[`Qwen/Qwen3-30B-A3B`](https://huggingface.co/Qwen/Qwen3-30B-A3B)(Apache-2.0,vanilla 版,2025-04 release)
- 训练框架:[`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL)
- Agent harness:[`camel-ai/camel`](https://github.com/camel-ai/camel) [`TerminalToolkit`](https://docs.camel-ai.org/key_modules/terminaltoolkit)
- Eval benchmark:[`laude-institute/terminal-bench`](https://github.com/laude-institute/terminal-bench)
- Rollout 训练数据 pool:[`camel-ai/seta`](https://github.com/camel-ai/seta) 1376 task
License: Apache-2.0(继承自 Qwen3-30B-A3B)。
如果你用这个 ckpt 做 paper / blog post,欢迎引用 [`HansBug/OpenClaw-RL`](https://github.com/HansBug/OpenClaw-RL) 仓库 + 这个 model card。

38
config.json Normal file
View File

@@ -0,0 +1,38 @@
{
"architectures": [
"Qwen3MoeForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"bos_token_id": 151643,
"decoder_sparse_step": 1,
"eos_token_id": 151645,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2048,
"initializer_range": 0.02,
"intermediate_size": 6144,
"max_position_embeddings": 40960,
"max_window_layers": 48,
"mlp_only_layers": [],
"model_type": "qwen3_moe",
"moe_intermediate_size": 768,
"norm_topk_prob": true,
"num_attention_heads": 32,
"num_experts": 128,
"num_experts_per_tok": 8,
"num_hidden_layers": 48,
"num_key_value_heads": 4,
"output_router_logits": false,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 1000000.0,
"router_aux_loss_coef": 0.001,
"sliding_window": null,
"tie_word_embeddings": false,
"torch_dtype": "bfloat16",
"transformers_version": "4.51.0",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

BIN
eval_pass_rate.png Normal file

Binary file not shown.

After

Width:  |  Height:  |  Size: 52 KiB

13
generation_config.json Normal file
View File

@@ -0,0 +1,13 @@
{
"bos_token_id": 151643,
"do_sample": true,
"eos_token_id": [
151645,
151643
],
"pad_token_id": 151643,
"temperature": 0.6,
"top_k": 20,
"top_p": 0.95,
"transformers_version": "4.51.0"
}

BIN
leaderboard.png Normal file

Binary file not shown.

After

Width:  |  Height:  |  Size: 46 KiB

151388
merges.txt Normal file

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d87654f409ae0c2daf1bbf8d52aad7796f973de28e39385dab5bece8f80df646
size 5368403824

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2ace3b84fae9d386cde1bfeafbff6643ab4821db2697f6c04ac2126e8b1988a6
size 5366338784

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:dcad34d518774682dcd0cc69ef11961547872e2abe6457d9fcae85a25f288b8e
size 5365807096

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0e1e77f2cb4f880cb55e383d24bf8946e8704c25be3ccbccb9f89007c083cfca
size 5365807912

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:773de90228818b49c2e90107c7f571a8a6bb4352640cfcec34de7f2340a9ec97
size 5366340528

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d3dd1923ee40bceda9c58cd1f4c29af7fe9bd5c64e059915e2189dbd8e532ce0
size 5365807832

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0e44b393301da6a6e77ea28bd5edacbb8a5255999f1fabb828b5b71c823274f0
size 5365807888

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:034830e720f81bd8756538629da33b20b377f64a4463205c06028d040819b771
size 5365807968

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ddf4a72a4efd9bea144694b509598a312280e5ada55dedd5b2d54611bc38c50d
size 5366340408

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:26e9bb6013595576fe90d0316b3a33d4ccfb8d02acb4c39d3c3d9b0cb65aba89
size 5365807856

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ad77a88f4bc90b14112bc51315360faa6c1dd75ffa9fe75339abeefaadfa95b1
size 5365807960

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e0727dee24c5c05100124a647c1fb746a041c1a95db56c916e5b32394a2c921c
size 2038500104

18874
model.safetensors.index.json Normal file

File diff suppressed because it is too large Load Diff

BIN
per_task_heatmap.png Normal file

Binary file not shown.

After

Width:  |  Height:  |  Size: 81 KiB

BIN
tokenizer.json (Stored with Git LFS) Normal file

Binary file not shown.

239
tokenizer_config.json Normal file
View File

@@ -0,0 +1,239 @@
{
"add_bos_token": false,
"add_prefix_space": false,
"added_tokens_decoder": {
"151643": {
"content": "<|endoftext|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151644": {
"content": "<|im_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151645": {
"content": "<|im_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151646": {
"content": "<|object_ref_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151647": {
"content": "<|object_ref_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151648": {
"content": "<|box_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151649": {
"content": "<|box_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151650": {
"content": "<|quad_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151651": {
"content": "<|quad_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151652": {
"content": "<|vision_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151653": {
"content": "<|vision_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151654": {
"content": "<|vision_pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151655": {
"content": "<|image_pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151656": {
"content": "<|video_pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"151657": {
"content": "<tool_call>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151658": {
"content": "</tool_call>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151659": {
"content": "<|fim_prefix|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151660": {
"content": "<|fim_middle|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151661": {
"content": "<|fim_suffix|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151662": {
"content": "<|fim_pad|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151663": {
"content": "<|repo_name|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151664": {
"content": "<|file_sep|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151665": {
"content": "<tool_response>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151666": {
"content": "</tool_response>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151667": {
"content": "<think>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
},
"151668": {
"content": "</think>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
}
},
"additional_special_tokens": [
"<|im_start|>",
"<|im_end|>",
"<|object_ref_start|>",
"<|object_ref_end|>",
"<|box_start|>",
"<|box_end|>",
"<|quad_start|>",
"<|quad_end|>",
"<|vision_start|>",
"<|vision_end|>",
"<|vision_pad|>",
"<|image_pad|>",
"<|video_pad|>"
],
"bos_token": null,
"chat_template": "{%- if tools %}\n {{- '<|im_start|>system\\n' }}\n {%- if messages[0].role == 'system' %}\n {{- messages[0].content + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {{- '<|im_start|>system\\n' + messages[0].content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if message.content is string %}\n {%- set content = message.content %}\n {%- else %}\n {%- set content = '' %}\n {%- endif %}\n {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- if loop.index0 > ns.last_query_index %}\n {%- if loop.last or (not loop.last and reasoning_content) %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content.strip('\\n') + '\\n</think>\\n\\n' + content.lstrip('\\n') }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls %}\n {%- for tool_call in message.tool_calls %}\n {%- if (loop.first and content) or (not loop.first) %}\n {{- '\\n' }}\n {%- endif %}\n {%- if tool_call.function %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {{- '<tool_call>\\n{\"name\": \"' }}\n {{- tool_call.name }}\n {{- '\", \"arguments\": ' }}\n {%- if tool_call.arguments is string %}\n {{- tool_call.arguments }}\n {%- else %}\n {{- tool_call.arguments | tojson }}\n {%- endif %}\n {{- '}\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.first or (messages[loop.index0 - 1].role != \"tool\") %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- endif %}\n{%- endif %}",
"clean_up_tokenization_spaces": false,
"eos_token": "<|im_end|>",
"errors": "replace",
"model_max_length": 131072,
"pad_token": "<|endoftext|>",
"split_special_tokens": false,
"tokenizer_class": "Qwen2Tokenizer",
"unk_token": null
}

3
training_curves.png Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ae8651c73c689c25cbf5ea9014c0e9673ecaac6d5463ad32ec173574d0485398
size 391498

1
vocab.json Normal file

File diff suppressed because one or more lines are too long